obu.c:clz:
  186|  5.61k|static inline int clz(const unsigned int mask) {
  187|  5.61k|    return __builtin_clz(mask);
  188|  5.61k|}
decode.c:ctz:
  182|   110k|static inline int ctz(const unsigned int mask) {
  183|   110k|    return __builtin_ctz(mask);
  184|   110k|}
decode.c:clz:
  186|  7.05M|static inline int clz(const unsigned int mask) {
  187|  7.05M|    return __builtin_clz(mask);
  188|  7.05M|}
getbits.c:clz:
  186|   133k|static inline int clz(const unsigned int mask) {
  187|   133k|    return __builtin_clz(mask);
  188|   133k|}
lf_mask.c:clz:
  186|  2.03M|static inline int clz(const unsigned int mask) {
  187|  2.03M|    return __builtin_clz(mask);
  188|  2.03M|}
warpmv.c:clz:
  186|  80.9k|static inline int clz(const unsigned int mask) {
  187|  80.9k|    return __builtin_clz(mask);
  188|  80.9k|}
warpmv.c:clzll:
  190|  76.3k|static inline int clzll(const unsigned long long mask) {
  191|  76.3k|    return __builtin_clzll(mask);
  192|  76.3k|}
looprestoration_tmpl.c:clz:
  186|   465k|static inline int clz(const unsigned int mask) {
  187|   465k|    return __builtin_clz(mask);
  188|   465k|}
recon_tmpl.c:clz:
  186|  17.4M|static inline int clz(const unsigned int mask) {
  187|  17.4M|    return __builtin_clz(mask);
  188|  17.4M|}
cdef_apply_tmpl.c:clz:
  186|   470k|static inline int clz(const unsigned int mask) {
  187|   470k|    return __builtin_clz(mask);
  188|   470k|}
ipred_prepare_tmpl.c:clz:
  186|  7.10M|static inline int clz(const unsigned int mask) {
  187|  7.10M|    return __builtin_clz(mask);
  188|  7.10M|}

fg_apply_tmpl.c:PXSTRIDE:
   79|  43.0k|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  43.0k|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 43.0k, False: 0]
  ------------------
   81|  43.0k|    return x >> 1;
   82|  43.0k|}
itx_tmpl.c:PXSTRIDE:
   79|  6.12M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  6.12M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 6.12M, False: 0]
  ------------------
   81|  6.12M|    return x >> 1;
   82|  6.12M|}
looprestoration_tmpl.c:PXSTRIDE:
   79|  1.20M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  1.20M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 1.20M, False: 0]
  ------------------
   81|  1.20M|    return x >> 1;
   82|  1.20M|}
recon_tmpl.c:PXSTRIDE:
   79|  15.2M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  15.2M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 15.2M, False: 0]
  ------------------
   81|  15.2M|    return x >> 1;
   82|  15.2M|}
cdef_apply_tmpl.c:PXSTRIDE:
   79|  4.27M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  4.27M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 4.27M, False: 0]
  ------------------
   81|  4.27M|    return x >> 1;
   82|  4.27M|}
ipred_prepare_tmpl.c:PXSTRIDE:
   79|  83.3M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  83.3M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 83.3M, False: 0]
  ------------------
   81|  83.3M|    return x >> 1;
   82|  83.3M|}
ipred_prepare_tmpl.c:pixel_set:
   66|  1.28M|static inline void pixel_set(pixel *const dst, const int val, const int num) {
   67|  20.9M|    for (int n = 0; n < num; n++)
  ------------------
  |  Branch (67:21): [True: 19.6M, False: 1.28M]
  ------------------
   68|  19.6M|        dst[n] = val;
   69|  1.28M|}
lf_apply_tmpl.c:PXSTRIDE:
   79|  3.10M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  3.10M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 3.10M, False: 0]
  ------------------
   81|  3.10M|    return x >> 1;
   82|  3.10M|}
lr_apply_tmpl.c:PXSTRIDE:
   79|   392k|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|   392k|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 392k, False: 0]
  ------------------
   81|   392k|    return x >> 1;
   82|   392k|}

lib.c:umin:
   47|  10.2k|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|  10.2k|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 0, False: 10.2k]
  ------------------
   49|  10.2k|}
obu.c:ulog2:
   67|  5.61k|static inline int ulog2(const unsigned v) {
   68|  5.61k|    return 31 ^ clz(v);
   69|  5.61k|}
obu.c:imin:
   39|   327k|static inline int imin(const int a, const int b) {
   40|   327k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 235k, False: 92.1k]
  ------------------
   41|   327k|}
obu.c:imax:
   35|   232k|static inline int imax(const int a, const int b) {
   36|   232k|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 20.5k, False: 211k]
  ------------------
   37|   232k|}
obu.c:iclip_u8:
   55|   102k|static inline int iclip_u8(const int v) {
   56|   102k|    return iclip(v, 0, 255);
   57|   102k|}
obu.c:iclip:
   51|   102k|static inline int iclip(const int v, const int min, const int max) {
   52|   102k|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 1.40k, False: 101k]
  |  Branch (52:28): [True: 1.12k, False: 100k]
  ------------------
   53|   102k|}
refmvs.c:imin:
   39|  21.8M|static inline int imin(const int a, const int b) {
   40|  21.8M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 10.3M, False: 11.5M]
  ------------------
   41|  21.8M|}
refmvs.c:apply_sign:
   59|   449k|static inline int apply_sign(const int v, const int s) {
   60|   449k|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 181k, False: 268k]
  ------------------
   61|   449k|}
refmvs.c:imax:
   35|  10.7M|static inline int imax(const int a, const int b) {
   36|  10.7M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 1.88M, False: 8.89M]
  ------------------
   37|  10.7M|}
refmvs.c:iclip:
   51|  9.97M|static inline int iclip(const int v, const int min, const int max) {
   52|  9.97M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 238k, False: 9.73M]
  |  Branch (52:28): [True: 217k, False: 9.51M]
  ------------------
   53|  9.97M|}
wedge.c:imax:
   35|    256|static inline int imax(const int a, const int b) {
   36|    256|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 128, False: 128]
  ------------------
   37|    256|}
wedge.c:imin:
   39|  2.48k|static inline int imin(const int a, const int b) {
   40|  2.48k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 1.41k, False: 1.06k]
  ------------------
   41|  2.48k|}
fg_apply_tmpl.c:imin:
   39|  30.4k|static inline int imin(const int a, const int b) {
   40|  30.4k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 5.36k, False: 25.0k]
  ------------------
   41|  30.4k|}
cdf.c:imin:
   39|   161k|static inline int imin(const int a, const int b) {
   40|   161k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 40.2k, False: 120k]
  ------------------
   41|   161k|}
decode.c:iclip:
   51|  1.19M|static inline int iclip(const int v, const int min, const int max) {
   52|  1.19M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 60.5k, False: 1.12M]
  |  Branch (52:28): [True: 23.7k, False: 1.10M]
  ------------------
   53|  1.19M|}
decode.c:apply_sign:
   59|   131k|static inline int apply_sign(const int v, const int s) {
   60|   131k|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 54.0k, False: 77.8k]
  ------------------
   61|   131k|}
decode.c:ulog2:
   67|  7.05M|static inline int ulog2(const unsigned v) {
   68|  7.05M|    return 31 ^ clz(v);
   69|  7.05M|}
decode.c:imax:
   35|  6.96M|static inline int imax(const int a, const int b) {
   36|  6.96M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 2.63M, False: 4.33M]
  ------------------
   37|  6.96M|}
decode.c:imin:
   39|  16.7M|static inline int imin(const int a, const int b) {
   40|  16.7M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 10.1M, False: 6.61M]
  ------------------
   41|  16.7M|}
decode.c:iclip_u8:
   55|   842k|static inline int iclip_u8(const int v) {
   56|   842k|    return iclip(v, 0, 255);
   57|   842k|}
getbits.c:ulog2:
   67|   133k|static inline int ulog2(const unsigned v) {
   68|   133k|    return 31 ^ clz(v);
   69|   133k|}
getbits.c:inv_recenter:
   75|  39.0k|static inline unsigned inv_recenter(const unsigned r, const unsigned v) {
   76|  39.0k|    if (v > (r << 1))
  ------------------
  |  Branch (76:9): [True: 2.33k, False: 36.7k]
  ------------------
   77|  2.33k|        return v;
   78|  36.7k|    else if ((v & 1) == 0)
  ------------------
  |  Branch (78:14): [True: 23.1k, False: 13.6k]
  ------------------
   79|  23.1k|        return (v >> 1) + r;
   80|  13.6k|    else
   81|  13.6k|        return r - ((v + 1) >> 1);
   82|  39.0k|}
lf_mask.c:imin:
   39|  27.1M|static inline int imin(const int a, const int b) {
   40|  27.1M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 2.73M, False: 24.4M]
  ------------------
   41|  27.1M|}
lf_mask.c:ulog2:
   67|  2.03M|static inline int ulog2(const unsigned v) {
   68|  2.03M|    return 31 ^ clz(v);
   69|  2.03M|}
lf_mask.c:imax:
   35|  1.02M|static inline int imax(const int a, const int b) {
   36|  1.02M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 967k, False: 61.6k]
  ------------------
   37|  1.02M|}
lf_mask.c:iclip:
   51|  3.16M|static inline int iclip(const int v, const int min, const int max) {
   52|  3.16M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 254k, False: 2.91M]
  |  Branch (52:28): [True: 68.9k, False: 2.84M]
  ------------------
   53|  3.16M|}
msac.c:inv_recenter:
   75|   133k|static inline unsigned inv_recenter(const unsigned r, const unsigned v) {
   76|   133k|    if (v > (r << 1))
  ------------------
  |  Branch (76:9): [True: 32.9k, False: 100k]
  ------------------
   77|  32.9k|        return v;
   78|   100k|    else if ((v & 1) == 0)
  ------------------
  |  Branch (78:14): [True: 47.7k, False: 53.1k]
  ------------------
   79|  47.7k|        return (v >> 1) + r;
   80|  53.1k|    else
   81|  53.1k|        return r - ((v + 1) >> 1);
   82|   133k|}
warpmv.c:apply_sign:
   59|   404k|static inline int apply_sign(const int v, const int s) {
   60|   404k|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 85.1k, False: 319k]
  ------------------
   61|   404k|}
warpmv.c:ulog2:
   67|  80.9k|static inline int ulog2(const unsigned v) {
   68|  80.9k|    return 31 ^ clz(v);
   69|  80.9k|}
warpmv.c:apply_sign64:
   63|   543k|static inline int apply_sign64(const int v, const int64_t s) {
   64|   543k|    return s < 0 ? -v : v;
  ------------------
  |  Branch (64:12): [True: 50.8k, False: 493k]
  ------------------
   65|   543k|}
warpmv.c:iclip:
   51|   782k|static inline int iclip(const int v, const int min, const int max) {
   52|   782k|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 16.7k, False: 765k]
  |  Branch (52:28): [True: 22.0k, False: 743k]
  ------------------
   53|   782k|}
warpmv.c:u64log2:
   71|  76.3k|static inline int u64log2(const uint64_t v) {
   72|  76.3k|    return 63 ^ clzll(v);
   73|  76.3k|}
itx_tmpl.c:iclip:
   51|   410M|static inline int iclip(const int v, const int min, const int max) {
   52|   410M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 15.5M, False: 395M]
  |  Branch (52:28): [True: 16.5M, False: 378M]
  ------------------
   53|   410M|}
itx_tmpl.c:imin:
   39|   246k|static inline int imin(const int a, const int b) {
   40|   246k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 48.2k, False: 198k]
  ------------------
   41|   246k|}
looprestoration_tmpl.c:iclip:
   51|  47.2M|static inline int iclip(const int v, const int min, const int max) {
   52|  47.2M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 982, False: 47.2M]
  |  Branch (52:28): [True: 893, False: 47.2M]
  ------------------
   53|  47.2M|}
looprestoration_tmpl.c:imax:
   35|  50.6M|static inline int imax(const int a, const int b) {
   36|  50.6M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 3.87M, False: 46.8M]
  ------------------
   37|  50.6M|}
looprestoration_tmpl.c:umin:
   47|  50.6M|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|  50.6M|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 50.6M, False: 41.8k]
  ------------------
   49|  50.6M|}
recon_tmpl.c:ulog2:
   67|  17.4M|static inline int ulog2(const unsigned v) {
   68|  17.4M|    return 31 ^ clz(v);
   69|  17.4M|}
recon_tmpl.c:imin:
   39|  55.1M|static inline int imin(const int a, const int b) {
   40|  55.1M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 46.0M, False: 9.12M]
  ------------------
   41|  55.1M|}
recon_tmpl.c:imax:
   35|  3.99M|static inline int imax(const int a, const int b) {
   36|  3.99M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 2.31M, False: 1.68M]
  ------------------
   37|  3.99M|}
recon_tmpl.c:umin:
   47|   171M|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|   171M|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 99.2M, False: 71.9M]
  ------------------
   49|   171M|}
recon_tmpl.c:apply_sign64:
   63|  2.56M|static inline int apply_sign64(const int v, const int64_t s) {
   64|  2.56M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (64:12): [True: 292k, False: 2.27M]
  ------------------
   65|  2.56M|}
recon_tmpl.c:iclip:
   51|   657k|static inline int iclip(const int v, const int min, const int max) {
   52|   657k|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 48.0k, False: 609k]
  |  Branch (52:28): [True: 9.53k, False: 600k]
  ------------------
   53|   657k|}
itx_1d.c:iclip:
   51|  1.04G|static inline int iclip(const int v, const int min, const int max) {
   52|  1.04G|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 8.43M, False: 1.03G]
  |  Branch (52:28): [True: 8.32M, False: 1.02G]
  ------------------
   53|  1.04G|}
scan.c:imax:
   35|  3.34k|static inline int imax(const int a, const int b) {
   36|  3.34k|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 2.82k, False: 523]
  ------------------
   37|  3.34k|}
cdef_apply_tmpl.c:imin:
   39|  1.99M|static inline int imin(const int a, const int b) {
   40|  1.99M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 1.71M, False: 285k]
  ------------------
   41|  1.99M|}
cdef_apply_tmpl.c:ulog2:
   67|   470k|static inline int ulog2(const unsigned v) {
   68|   470k|    return 31 ^ clz(v);
   69|   470k|}
ipred_prepare_tmpl.c:imin:
   39|  23.1M|static inline int imin(const int a, const int b) {
   40|  23.1M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 20.4M, False: 2.76M]
  ------------------
   41|  23.1M|}
lf_apply_tmpl.c:imin:
   39|   728k|static inline int imin(const int a, const int b) {
   40|   728k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 211k, False: 517k]
  ------------------
   41|   728k|}
lr_apply_tmpl.c:imin:
   39|  95.5k|static inline int imin(const int a, const int b) {
   40|  95.5k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 41.2k, False: 54.3k]
  ------------------
   41|  95.5k|}

dav1d_cdef_brow_8bpc:
  102|  66.2k|{
  103|  66.2k|    Dav1dFrameContext *const f = (Dav1dFrameContext *)tc->f;
  104|  66.2k|    const int bitdepth_min_8 = BITDEPTH == 8 ? 0 : f->cur.p.bpc - 8;
  ------------------
  |  Branch (104:32): [True: 66.2k, Folded]
  ------------------
  105|  66.2k|    const Dav1dDSPContext *const dsp = f->dsp;
  106|  66.2k|    enum CdefEdgeFlags edges = CDEF_HAVE_BOTTOM | (by_start > 0 ? CDEF_HAVE_TOP : 0);
  ------------------
  |  Branch (106:52): [True: 63.3k, False: 2.97k]
  ------------------
  107|  66.2k|    pixel *ptrs[3] = { p[0], p[1], p[2] };
  108|  66.2k|    const int sbsz = 16;
  109|  66.2k|    const int sb64w = f->sb128w << 1;
  110|  66.2k|    const int damping = f->frame_hdr->cdef.damping + bitdepth_min_8;
  111|  66.2k|    const enum Dav1dPixelLayout layout = f->cur.p.layout;
  112|  66.2k|    const int uv_idx = DAV1D_PIXEL_LAYOUT_I444 - layout;
  113|  66.2k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  114|  66.2k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  115|  66.2k|    static const uint8_t uv_dirs[2][8] = { { 0, 1, 2, 3, 4, 5, 6, 7 },
  116|  66.2k|                                           { 7, 0, 2, 4, 5, 6, 6, 6 } };
  117|  66.2k|    const uint8_t *uv_dir = uv_dirs[layout == DAV1D_PIXEL_LAYOUT_I422];
  118|  66.2k|    const int have_tt = f->c->n_tc > 1;
  119|  66.2k|    const int sb128 = f->seq_hdr->sb128;
  120|  66.2k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  121|  66.2k|    const ptrdiff_t y_stride = PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|  66.2k|#define PXSTRIDE(x) (x)
  ------------------
  122|  66.2k|    const ptrdiff_t uv_stride = PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  66.2k|#define PXSTRIDE(x) (x)
  ------------------
  123|       |
  124|   388k|    for (int bit = 0, by = by_start; by < by_end; by += 2, edges |= CDEF_HAVE_TOP) {
  ------------------
  |  Branch (124:38): [True: 322k, False: 66.2k]
  ------------------
  125|   322k|        const int tf = tc->top_pre_cdef_toggle;
  126|   322k|        const int by_idx = (by & 30) >> 1;
  127|   322k|        if (by + 2 >= f->bh) edges &= ~CDEF_HAVE_BOTTOM;
  ------------------
  |  Branch (127:13): [True: 2.75k, False: 319k]
  ------------------
  128|       |
  129|   322k|        if ((!have_tt || sbrow_start || by + 2 < by_end) &&
  ------------------
  |  Branch (129:14): [True: 322k, False: 0]
  |  Branch (129:26): [True: 0, False: 0]
  |  Branch (129:41): [True: 0, False: 0]
  ------------------
  130|   322k|            edges & CDEF_HAVE_BOTTOM)
  ------------------
  |  Branch (130:13): [True: 319k, False: 2.75k]
  ------------------
  131|   319k|        {
  132|       |            // backup pre-filter data for next iteration
  133|   319k|            pixel *const cdef_top_bak[3] = {
  134|   319k|                f->lf.cdef_line[!tf][0] + have_tt * sby * 4 * y_stride,
  135|   319k|                f->lf.cdef_line[!tf][1] + have_tt * sby * 8 * uv_stride,
  136|   319k|                f->lf.cdef_line[!tf][2] + have_tt * sby * 8 * uv_stride
  137|   319k|            };
  138|   319k|            backup2lines(cdef_top_bak, ptrs, f->cur.stride, layout);
  139|   319k|        }
  140|       |
  141|   322k|        ALIGN_STK_16(pixel, lr_bak, 2 /* idx */, [3 /* plane */][8 /* y */][2 /* x */]);
  ------------------
  |  |  100|   322k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   322k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  142|   322k|        pixel *iptrs[3] = { ptrs[0], ptrs[1], ptrs[2] };
  143|   322k|        edges &= ~CDEF_HAVE_LEFT;
  144|   322k|        edges |= CDEF_HAVE_RIGHT;
  145|   322k|        enum Backup2x8Flags prev_flag = 0;
  146|  1.21M|        for (int sbx = 0; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (146:27): [True: 896k, False: 322k]
  ------------------
  147|   896k|            const int sb128x = sbx >> 1;
  148|   896k|            const int sb64_idx = ((by & sbsz) >> 3) + (sbx & 1);
  149|   896k|            const int cdef_idx = lflvl[sb128x].cdef_idx[sb64_idx];
  150|   896k|            if (cdef_idx == -1 ||
  ------------------
  |  Branch (150:17): [True: 645k, False: 250k]
  ------------------
  151|   250k|                (!f->frame_hdr->cdef.y_strength[cdef_idx] &&
  ------------------
  |  Branch (151:18): [True: 122k, False: 127k]
  ------------------
  152|   122k|                 !f->frame_hdr->cdef.uv_strength[cdef_idx]))
  ------------------
  |  Branch (152:18): [True: 115k, False: 7.40k]
  ------------------
  153|   761k|            {
  154|   761k|                prev_flag = 0;
  155|   761k|                goto next_sb;
  156|   761k|            }
  157|       |
  158|       |            // Create a complete 32-bit mask for the sb row ahead of time.
  159|   135k|            const uint16_t (*noskip_row)[2] = &lflvl[sb128x].noskip_mask[by_idx];
  160|   135k|            const unsigned noskip_mask = (unsigned) noskip_row[0][1] << 16 |
  161|   135k|                                                    noskip_row[0][0];
  162|       |
  163|   135k|            const int y_lvl = f->frame_hdr->cdef.y_strength[cdef_idx];
  164|   135k|            const int uv_lvl = f->frame_hdr->cdef.uv_strength[cdef_idx];
  165|   135k|            const enum Backup2x8Flags flag = !!y_lvl + (!!uv_lvl << 1);
  166|       |
  167|   135k|            const int y_pri_lvl = (y_lvl >> 2) << bitdepth_min_8;
  168|   135k|            int y_sec_lvl = y_lvl & 3;
  169|   135k|            y_sec_lvl += y_sec_lvl == 3;
  170|   135k|            y_sec_lvl <<= bitdepth_min_8;
  171|       |
  172|   135k|            const int uv_pri_lvl = (uv_lvl >> 2) << bitdepth_min_8;
  173|   135k|            int uv_sec_lvl = uv_lvl & 3;
  174|   135k|            uv_sec_lvl += uv_sec_lvl == 3;
  175|   135k|            uv_sec_lvl <<= bitdepth_min_8;
  176|       |
  177|   135k|            pixel *bptrs[3] = { iptrs[0], iptrs[1], iptrs[2] };
  178|  1.12M|            for (int bx = sbx * sbsz; bx < imin((sbx + 1) * sbsz, f->bw);
  ------------------
  |  Branch (178:39): [True: 991k, False: 135k]
  ------------------
  179|   991k|                 bx += 2, edges |= CDEF_HAVE_LEFT)
  180|   991k|            {
  181|   991k|                if (bx + 2 >= f->bw) edges &= ~CDEF_HAVE_RIGHT;
  ------------------
  |  Branch (181:21): [True: 20.9k, False: 970k]
  ------------------
  182|       |
  183|       |                // check if this 8x8 block had any coded coefficients; if not,
  184|       |                // go to the next block
  185|   991k|                const uint32_t bx_mask = 3U << (bx & 30);
  186|   991k|                if (!(noskip_mask & bx_mask)) {
  ------------------
  |  Branch (186:21): [True: 110k, False: 881k]
  ------------------
  187|   110k|                    prev_flag = 0;
  188|   110k|                    goto next_b;
  189|   110k|                }
  190|   881k|                const enum Backup2x8Flags do_left = (prev_flag ^ flag) & flag;
  191|   881k|                prev_flag = flag;
  192|   881k|                if (do_left && edges & CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (192:21): [True: 43.8k, False: 837k]
  |  Branch (192:32): [True: 24.8k, False: 18.9k]
  ------------------
  193|       |                    // we didn't backup the prefilter data because it wasn't
  194|       |                    // there, so do it here instead
  195|  24.8k|                    backup2x8(lr_bak[bit], bptrs, f->cur.stride, 0, layout, do_left);
  196|  24.8k|                }
  197|   881k|                if (edges & CDEF_HAVE_RIGHT) {
  ------------------
  |  Branch (197:21): [True: 862k, False: 18.6k]
  ------------------
  198|       |                    // backup pre-filter data for next iteration
  199|   862k|                    backup2x8(lr_bak[!bit], bptrs, f->cur.stride, 8, layout, flag);
  200|   862k|                }
  201|       |
  202|   881k|                int dir;
  203|   881k|                unsigned variance;
  204|   881k|                if (y_pri_lvl || uv_pri_lvl)
  ------------------
  |  Branch (204:21): [True: 758k, False: 122k]
  |  Branch (204:34): [True: 83.6k, False: 38.8k]
  ------------------
  205|   842k|                    dir = dsp->cdef.dir(bptrs[0], f->cur.stride[0],
  206|   842k|                                        &variance HIGHBD_CALL_SUFFIX);
  207|       |
  208|   881k|                const pixel *top, *bot;
  209|   881k|                ptrdiff_t offset;
  210|       |
  211|   881k|                if (!have_tt) goto st_y;
  ------------------
  |  Branch (211:21): [True: 881k, False: 0]
  ------------------
  212|      0|                if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (212:21): [True: 0, False: 0]
  |  Branch (212:36): [True: 0, False: 0]
  ------------------
  213|      0|                    if (resize) {
  ------------------
  |  Branch (213:25): [True: 0, False: 0]
  ------------------
  214|      0|                        offset = (sby - 1) * 4 * y_stride + bx * 4;
  215|      0|                        top = &f->lf.cdef_lpf_line[0][offset];
  216|      0|                    } else {
  217|      0|                        offset = (sby * (4 << sb128) - 4) * y_stride + bx * 4;
  218|      0|                        top = &f->lf.lr_lpf_line[0][offset];
  219|      0|                    }
  220|      0|                    bot = bptrs[0] + 8 * y_stride;
  221|      0|                } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (221:28): [True: 0, False: 0]
  |  Branch (221:44): [True: 0, False: 0]
  ------------------
  222|      0|                    top = &f->lf.cdef_line[tf][0][sby * 4 * y_stride + bx * 4];
  223|      0|                    if (resize) {
  ------------------
  |  Branch (223:25): [True: 0, False: 0]
  ------------------
  224|      0|                        offset = (sby * 4 + 2) * y_stride + bx * 4;
  225|      0|                        bot = &f->lf.cdef_lpf_line[0][offset];
  226|      0|                    } else {
  227|      0|                        const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  228|      0|                        offset = line * y_stride + bx * 4;
  229|      0|                        bot = &f->lf.lr_lpf_line[0][offset];
  230|      0|                    }
  231|      0|                } else {
  232|   881k|            st_y:;
  233|   881k|                    offset = sby * 4 * y_stride;
  234|   881k|                    top = &f->lf.cdef_line[tf][0][have_tt * offset + bx * 4];
  235|   881k|                    bot = bptrs[0] + 8 * y_stride;
  236|   881k|                }
  237|   881k|                if (y_pri_lvl) {
  ------------------
  |  Branch (237:21): [True: 758k, False: 122k]
  ------------------
  238|   758k|                    const int adj_y_pri_lvl = adjust_strength(y_pri_lvl, variance);
  239|   758k|                    if (adj_y_pri_lvl || y_sec_lvl)
  ------------------
  |  Branch (239:25): [True: 516k, False: 242k]
  |  Branch (239:42): [True: 166k, False: 75.5k]
  ------------------
  240|   682k|                        dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  241|   682k|                                        top, bot, adj_y_pri_lvl, y_sec_lvl,
  242|   682k|                                        dir, damping, edges HIGHBD_CALL_SUFFIX);
  243|   758k|                } else if (y_sec_lvl)
  ------------------
  |  Branch (243:28): [True: 78.7k, False: 43.7k]
  ------------------
  244|  78.7k|                    dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  245|  78.7k|                                    top, bot, 0, y_sec_lvl, 0, damping,
  246|  78.7k|                                    edges HIGHBD_CALL_SUFFIX);
  247|       |
  248|   881k|                if (!uv_lvl) goto skip_uv;
  ------------------
  |  Branch (248:21): [True: 101k, False: 779k]
  ------------------
  249|   881k|                assert(layout != DAV1D_PIXEL_LAYOUT_I400);
  ------------------
  |  Branch (249:17): [True: 779k, False: 0]
  ------------------
  250|       |
  251|   779k|                const int uvdir = uv_pri_lvl ? uv_dir[dir] : 0;
  ------------------
  |  Branch (251:35): [True: 753k, False: 26.3k]
  ------------------
  252|  2.33M|                for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (252:34): [True: 1.55M, False: 779k]
  ------------------
  253|  1.55M|                    if (!have_tt) goto st_uv;
  ------------------
  |  Branch (253:25): [True: 1.55M, False: 0]
  ------------------
  254|      0|                    if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (254:25): [True: 0, False: 0]
  |  Branch (254:40): [True: 0, False: 0]
  ------------------
  255|      0|                        if (resize) {
  ------------------
  |  Branch (255:29): [True: 0, False: 0]
  ------------------
  256|      0|                            offset = (sby - 1) * 4 * uv_stride + (bx * 4 >> ss_hor);
  257|      0|                            top = &f->lf.cdef_lpf_line[pl][offset];
  258|      0|                        } else {
  259|      0|                            const int line = sby * (4 << sb128) - 4;
  260|      0|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  261|      0|                            top = &f->lf.lr_lpf_line[pl][offset];
  262|      0|                        }
  263|      0|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  264|      0|                    } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (264:32): [True: 0, False: 0]
  |  Branch (264:48): [True: 0, False: 0]
  ------------------
  265|      0|                        const ptrdiff_t top_offset = sby * 8 * uv_stride +
  266|      0|                                                     (bx * 4 >> ss_hor);
  267|      0|                        top = &f->lf.cdef_line[tf][pl][top_offset];
  268|      0|                        if (resize) {
  ------------------
  |  Branch (268:29): [True: 0, False: 0]
  ------------------
  269|      0|                            offset = (sby * 4 + 2) * uv_stride + (bx * 4 >> ss_hor);
  270|      0|                            bot = &f->lf.cdef_lpf_line[pl][offset];
  271|      0|                        } else {
  272|      0|                            const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  273|      0|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  274|      0|                            bot = &f->lf.lr_lpf_line[pl][offset];
  275|      0|                        }
  276|      0|                    } else {
  277|  1.55M|                st_uv:;
  278|  1.55M|                        const ptrdiff_t offset = sby * 8 * uv_stride;
  279|  1.55M|                        top = &f->lf.cdef_line[tf][pl][have_tt * offset + (bx * 4 >> ss_hor)];
  280|  1.55M|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  281|  1.55M|                    }
  282|  1.55M|                    dsp->cdef.fb[uv_idx](bptrs[pl], f->cur.stride[1],
  283|  1.55M|                                         lr_bak[bit][pl], top, bot,
  284|  1.55M|                                         uv_pri_lvl, uv_sec_lvl, uvdir,
  285|  1.55M|                                         damping - 1, edges HIGHBD_CALL_SUFFIX);
  286|  1.55M|                }
  287|       |
  288|   881k|            skip_uv:
  289|   881k|                bit ^= 1;
  290|       |
  291|   991k|            next_b:
  292|   991k|                bptrs[0] += 8;
  293|   991k|                bptrs[1] += 8 >> ss_hor;
  294|   991k|                bptrs[2] += 8 >> ss_hor;
  295|   991k|            }
  296|       |
  297|   896k|        next_sb:
  298|   896k|            iptrs[0] += sbsz * 4;
  299|   896k|            iptrs[1] += sbsz * 4 >> ss_hor;
  300|   896k|            iptrs[2] += sbsz * 4 >> ss_hor;
  301|   896k|        }
  302|       |
  303|   322k|        ptrs[0] += 8 * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|   322k|#define PXSTRIDE(x) (x)
  ------------------
  304|   322k|        ptrs[1] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  ------------------
  |  |   53|   322k|#define PXSTRIDE(x) (x)
  ------------------
  305|   322k|        ptrs[2] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  ------------------
  |  |   53|   322k|#define PXSTRIDE(x) (x)
  ------------------
  306|   322k|        tc->top_pre_cdef_toggle ^= 1;
  307|   322k|    }
  308|  66.2k|}
cdef_apply_tmpl.c:backup2lines:
   44|   605k|{
   45|   605k|    const ptrdiff_t y_stride = PXSTRIDE(stride[0]);
  ------------------
  |  |   53|   605k|#define PXSTRIDE(x) (x)
  ------------------
   46|   605k|    if (y_stride < 0)
  ------------------
  |  Branch (46:9): [True: 0, False: 605k]
  ------------------
   47|      0|        pixel_copy(dst[0] + y_stride, src[0] + 7 * y_stride, -2 * y_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   48|   605k|    else
   49|   605k|        pixel_copy(dst[0], src[0] + 6 * y_stride, 2 * y_stride);
  ------------------
  |  |   47|   605k|#define pixel_copy memcpy
  ------------------
   50|       |
   51|   605k|    if (layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (51:9): [True: 345k, False: 259k]
  ------------------
   52|   345k|        const ptrdiff_t uv_stride = PXSTRIDE(stride[1]);
  ------------------
  |  |   53|   345k|#define PXSTRIDE(x) (x)
  ------------------
   53|   345k|        if (uv_stride < 0) {
  ------------------
  |  Branch (53:13): [True: 0, False: 345k]
  ------------------
   54|      0|            const int uv_off = layout == DAV1D_PIXEL_LAYOUT_I420 ? 3 : 7;
  ------------------
  |  Branch (54:32): [True: 0, False: 0]
  ------------------
   55|      0|            pixel_copy(dst[1] + uv_stride, src[1] + uv_off * uv_stride, -2 * uv_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   56|      0|            pixel_copy(dst[2] + uv_stride, src[2] + uv_off * uv_stride, -2 * uv_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   57|   345k|        } else {
   58|   345k|            const int uv_off = layout == DAV1D_PIXEL_LAYOUT_I420 ? 2 : 6;
  ------------------
  |  Branch (58:32): [True: 146k, False: 198k]
  ------------------
   59|   345k|            pixel_copy(dst[1], src[1] + uv_off * uv_stride, 2 * uv_stride);
  ------------------
  |  |   47|   345k|#define pixel_copy memcpy
  ------------------
   60|   345k|            pixel_copy(dst[2], src[2] + uv_off * uv_stride, 2 * uv_stride);
  ------------------
  |  |   47|   345k|#define pixel_copy memcpy
  ------------------
   61|   345k|        }
   62|   345k|    }
   63|   605k|}
cdef_apply_tmpl.c:backup2x8:
   70|  1.17M|{
   71|  1.17M|    ptrdiff_t y_off = 0;
   72|  1.17M|    if (flag & BACKUP_2X8_Y) {
  ------------------
  |  Branch (72:9): [True: 1.08M, False: 89.8k]
  ------------------
   73|  9.78M|        for (int y = 0; y < 8; y++, y_off += PXSTRIDE(src_stride[0]))
  ------------------
  |  |   53|  8.69M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (73:25): [True: 8.69M, False: 1.08M]
  ------------------
   74|  8.69M|            pixel_copy(dst[0][y], &src[0][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  8.69M|#define pixel_copy memcpy
  ------------------
   75|  1.08M|    }
   76|       |
   77|  1.17M|    if (layout == DAV1D_PIXEL_LAYOUT_I400 || !(flag & BACKUP_2X8_UV))
  ------------------
  |  Branch (77:9): [True: 128k, False: 1.04M]
  |  Branch (77:46): [True: 110k, False: 938k]
  ------------------
   78|   238k|        return;
   79|       |
   80|   938k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
   81|   938k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
   82|       |
   83|   938k|    x_off >>= ss_hor;
   84|   938k|    y_off = 0;
   85|  5.36M|    for (int y = 0; y < (8 >> ss_ver); y++, y_off += PXSTRIDE(src_stride[1])) {
  ------------------
  |  |   53|  4.43M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (85:21): [True: 4.43M, False: 938k]
  ------------------
   86|  4.43M|        pixel_copy(dst[1][y], &src[1][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  4.43M|#define pixel_copy memcpy
  ------------------
   87|  4.43M|        pixel_copy(dst[2][y], &src[2][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  4.43M|#define pixel_copy memcpy
  ------------------
   88|  4.43M|    }
   89|   938k|}
cdef_apply_tmpl.c:adjust_strength:
   91|   988k|static int adjust_strength(const int strength, const unsigned var) {
   92|   988k|    if (!var) return 0;
  ------------------
  |  Branch (92:9): [True: 427k, False: 560k]
  ------------------
   93|   560k|    const int i = var >> 6 ? imin(ulog2(var >> 6), 12) : 0;
  ------------------
  |  Branch (93:19): [True: 470k, False: 89.6k]
  ------------------
   94|   560k|    return (strength * (4 + i) + 8) >> 4;
   95|   988k|}
dav1d_cdef_brow_16bpc:
  102|  47.6k|{
  103|  47.6k|    Dav1dFrameContext *const f = (Dav1dFrameContext *)tc->f;
  104|  47.6k|    const int bitdepth_min_8 = BITDEPTH == 8 ? 0 : f->cur.p.bpc - 8;
  ------------------
  |  Branch (104:32): [Folded, False: 47.6k]
  ------------------
  105|  47.6k|    const Dav1dDSPContext *const dsp = f->dsp;
  106|  47.6k|    enum CdefEdgeFlags edges = CDEF_HAVE_BOTTOM | (by_start > 0 ? CDEF_HAVE_TOP : 0);
  ------------------
  |  Branch (106:52): [True: 45.3k, False: 2.32k]
  ------------------
  107|  47.6k|    pixel *ptrs[3] = { p[0], p[1], p[2] };
  108|  47.6k|    const int sbsz = 16;
  109|  47.6k|    const int sb64w = f->sb128w << 1;
  110|  47.6k|    const int damping = f->frame_hdr->cdef.damping + bitdepth_min_8;
  111|  47.6k|    const enum Dav1dPixelLayout layout = f->cur.p.layout;
  112|  47.6k|    const int uv_idx = DAV1D_PIXEL_LAYOUT_I444 - layout;
  113|  47.6k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  114|  47.6k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  115|  47.6k|    static const uint8_t uv_dirs[2][8] = { { 0, 1, 2, 3, 4, 5, 6, 7 },
  116|  47.6k|                                           { 7, 0, 2, 4, 5, 6, 6, 6 } };
  117|  47.6k|    const uint8_t *uv_dir = uv_dirs[layout == DAV1D_PIXEL_LAYOUT_I422];
  118|  47.6k|    const int have_tt = f->c->n_tc > 1;
  119|  47.6k|    const int sb128 = f->seq_hdr->sb128;
  120|  47.6k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  121|  47.6k|    const ptrdiff_t y_stride = PXSTRIDE(f->cur.stride[0]);
  122|  47.6k|    const ptrdiff_t uv_stride = PXSTRIDE(f->cur.stride[1]);
  123|       |
  124|   335k|    for (int bit = 0, by = by_start; by < by_end; by += 2, edges |= CDEF_HAVE_TOP) {
  ------------------
  |  Branch (124:38): [True: 287k, False: 47.6k]
  ------------------
  125|   287k|        const int tf = tc->top_pre_cdef_toggle;
  126|   287k|        const int by_idx = (by & 30) >> 1;
  127|   287k|        if (by + 2 >= f->bh) edges &= ~CDEF_HAVE_BOTTOM;
  ------------------
  |  Branch (127:13): [True: 2.08k, False: 285k]
  ------------------
  128|       |
  129|   287k|        if ((!have_tt || sbrow_start || by + 2 < by_end) &&
  ------------------
  |  Branch (129:14): [True: 287k, False: 0]
  |  Branch (129:26): [True: 0, False: 0]
  |  Branch (129:41): [True: 0, False: 0]
  ------------------
  130|   287k|            edges & CDEF_HAVE_BOTTOM)
  ------------------
  |  Branch (130:13): [True: 285k, False: 2.08k]
  ------------------
  131|   285k|        {
  132|       |            // backup pre-filter data for next iteration
  133|   285k|            pixel *const cdef_top_bak[3] = {
  134|   285k|                f->lf.cdef_line[!tf][0] + have_tt * sby * 4 * y_stride,
  135|   285k|                f->lf.cdef_line[!tf][1] + have_tt * sby * 8 * uv_stride,
  136|   285k|                f->lf.cdef_line[!tf][2] + have_tt * sby * 8 * uv_stride
  137|   285k|            };
  138|   285k|            backup2lines(cdef_top_bak, ptrs, f->cur.stride, layout);
  139|   285k|        }
  140|       |
  141|   287k|        ALIGN_STK_16(pixel, lr_bak, 2 /* idx */, [3 /* plane */][8 /* y */][2 /* x */]);
  ------------------
  |  |  100|   287k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   287k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  142|   287k|        pixel *iptrs[3] = { ptrs[0], ptrs[1], ptrs[2] };
  143|   287k|        edges &= ~CDEF_HAVE_LEFT;
  144|   287k|        edges |= CDEF_HAVE_RIGHT;
  145|   287k|        enum Backup2x8Flags prev_flag = 0;
  146|   969k|        for (int sbx = 0; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (146:27): [True: 681k, False: 287k]
  ------------------
  147|   681k|            const int sb128x = sbx >> 1;
  148|   681k|            const int sb64_idx = ((by & sbsz) >> 3) + (sbx & 1);
  149|   681k|            const int cdef_idx = lflvl[sb128x].cdef_idx[sb64_idx];
  150|   681k|            if (cdef_idx == -1 ||
  ------------------
  |  Branch (150:17): [True: 542k, False: 138k]
  ------------------
  151|   138k|                (!f->frame_hdr->cdef.y_strength[cdef_idx] &&
  ------------------
  |  Branch (151:18): [True: 83.3k, False: 55.6k]
  ------------------
  152|  83.3k|                 !f->frame_hdr->cdef.uv_strength[cdef_idx]))
  ------------------
  |  Branch (152:18): [True: 70.2k, False: 13.0k]
  ------------------
  153|   612k|            {
  154|   612k|                prev_flag = 0;
  155|   612k|                goto next_sb;
  156|   612k|            }
  157|       |
  158|       |            // Create a complete 32-bit mask for the sb row ahead of time.
  159|  68.7k|            const uint16_t (*noskip_row)[2] = &lflvl[sb128x].noskip_mask[by_idx];
  160|  68.7k|            const unsigned noskip_mask = (unsigned) noskip_row[0][1] << 16 |
  161|  68.7k|                                                    noskip_row[0][0];
  162|       |
  163|  68.7k|            const int y_lvl = f->frame_hdr->cdef.y_strength[cdef_idx];
  164|  68.7k|            const int uv_lvl = f->frame_hdr->cdef.uv_strength[cdef_idx];
  165|  68.7k|            const enum Backup2x8Flags flag = !!y_lvl + (!!uv_lvl << 1);
  166|       |
  167|  68.7k|            const int y_pri_lvl = (y_lvl >> 2) << bitdepth_min_8;
  168|  68.7k|            int y_sec_lvl = y_lvl & 3;
  169|  68.7k|            y_sec_lvl += y_sec_lvl == 3;
  170|  68.7k|            y_sec_lvl <<= bitdepth_min_8;
  171|       |
  172|  68.7k|            const int uv_pri_lvl = (uv_lvl >> 2) << bitdepth_min_8;
  173|  68.7k|            int uv_sec_lvl = uv_lvl & 3;
  174|  68.7k|            uv_sec_lvl += uv_sec_lvl == 3;
  175|  68.7k|            uv_sec_lvl <<= bitdepth_min_8;
  176|       |
  177|  68.7k|            pixel *bptrs[3] = { iptrs[0], iptrs[1], iptrs[2] };
  178|   397k|            for (int bx = sbx * sbsz; bx < imin((sbx + 1) * sbsz, f->bw);
  ------------------
  |  Branch (178:39): [True: 328k, False: 68.7k]
  ------------------
  179|   328k|                 bx += 2, edges |= CDEF_HAVE_LEFT)
  180|   328k|            {
  181|   328k|                if (bx + 2 >= f->bw) edges &= ~CDEF_HAVE_RIGHT;
  ------------------
  |  Branch (181:21): [True: 25.9k, False: 302k]
  ------------------
  182|       |
  183|       |                // check if this 8x8 block had any coded coefficients; if not,
  184|       |                // go to the next block
  185|   328k|                const uint32_t bx_mask = 3U << (bx & 30);
  186|   328k|                if (!(noskip_mask & bx_mask)) {
  ------------------
  |  Branch (186:21): [True: 15.6k, False: 313k]
  ------------------
  187|  15.6k|                    prev_flag = 0;
  188|  15.6k|                    goto next_b;
  189|  15.6k|                }
  190|   313k|                const enum Backup2x8Flags do_left = (prev_flag ^ flag) & flag;
  191|   313k|                prev_flag = flag;
  192|   313k|                if (do_left && edges & CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (192:21): [True: 27.0k, False: 286k]
  |  Branch (192:32): [True: 1.75k, False: 25.2k]
  ------------------
  193|       |                    // we didn't backup the prefilter data because it wasn't
  194|       |                    // there, so do it here instead
  195|  1.75k|                    backup2x8(lr_bak[bit], bptrs, f->cur.stride, 0, layout, do_left);
  196|  1.75k|                }
  197|   313k|                if (edges & CDEF_HAVE_RIGHT) {
  ------------------
  |  Branch (197:21): [True: 288k, False: 25.0k]
  ------------------
  198|       |                    // backup pre-filter data for next iteration
  199|   288k|                    backup2x8(lr_bak[!bit], bptrs, f->cur.stride, 8, layout, flag);
  200|   288k|                }
  201|       |
  202|   313k|                int dir;
  203|   313k|                unsigned variance;
  204|   313k|                if (y_pri_lvl || uv_pri_lvl)
  ------------------
  |  Branch (204:21): [True: 229k, False: 83.3k]
  |  Branch (204:34): [True: 58.4k, False: 24.8k]
  ------------------
  205|   288k|                    dir = dsp->cdef.dir(bptrs[0], f->cur.stride[0],
  206|   288k|                                        &variance HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   288k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  207|       |
  208|   313k|                const pixel *top, *bot;
  209|   313k|                ptrdiff_t offset;
  210|       |
  211|   313k|                if (!have_tt) goto st_y;
  ------------------
  |  Branch (211:21): [True: 313k, False: 0]
  ------------------
  212|      0|                if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (212:21): [True: 0, False: 0]
  |  Branch (212:36): [True: 0, False: 0]
  ------------------
  213|      0|                    if (resize) {
  ------------------
  |  Branch (213:25): [True: 0, False: 0]
  ------------------
  214|      0|                        offset = (sby - 1) * 4 * y_stride + bx * 4;
  215|      0|                        top = &f->lf.cdef_lpf_line[0][offset];
  216|      0|                    } else {
  217|      0|                        offset = (sby * (4 << sb128) - 4) * y_stride + bx * 4;
  218|      0|                        top = &f->lf.lr_lpf_line[0][offset];
  219|      0|                    }
  220|      0|                    bot = bptrs[0] + 8 * y_stride;
  221|      0|                } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (221:28): [True: 0, False: 0]
  |  Branch (221:44): [True: 0, False: 0]
  ------------------
  222|      0|                    top = &f->lf.cdef_line[tf][0][sby * 4 * y_stride + bx * 4];
  223|      0|                    if (resize) {
  ------------------
  |  Branch (223:25): [True: 0, False: 0]
  ------------------
  224|      0|                        offset = (sby * 4 + 2) * y_stride + bx * 4;
  225|      0|                        bot = &f->lf.cdef_lpf_line[0][offset];
  226|      0|                    } else {
  227|      0|                        const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  228|      0|                        offset = line * y_stride + bx * 4;
  229|      0|                        bot = &f->lf.lr_lpf_line[0][offset];
  230|      0|                    }
  231|      0|                } else {
  232|   313k|            st_y:;
  233|   313k|                    offset = sby * 4 * y_stride;
  234|   313k|                    top = &f->lf.cdef_line[tf][0][have_tt * offset + bx * 4];
  235|   313k|                    bot = bptrs[0] + 8 * y_stride;
  236|   313k|                }
  237|   313k|                if (y_pri_lvl) {
  ------------------
  |  Branch (237:21): [True: 229k, False: 83.3k]
  ------------------
  238|   229k|                    const int adj_y_pri_lvl = adjust_strength(y_pri_lvl, variance);
  239|   229k|                    if (adj_y_pri_lvl || y_sec_lvl)
  ------------------
  |  Branch (239:25): [True: 21.9k, False: 207k]
  |  Branch (239:42): [True: 130k, False: 77.7k]
  ------------------
  240|   152k|                        dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  241|   152k|                                        top, bot, adj_y_pri_lvl, y_sec_lvl,
  242|   152k|                                        dir, damping, edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   152k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  243|   229k|                } else if (y_sec_lvl)
  ------------------
  |  Branch (243:28): [True: 32.0k, False: 51.2k]
  ------------------
  244|  32.0k|                    dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  245|  32.0k|                                    top, bot, 0, y_sec_lvl, 0, damping,
  246|  32.0k|                                    edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  32.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  247|       |
  248|   313k|                if (!uv_lvl) goto skip_uv;
  ------------------
  |  Branch (248:21): [True: 145k, False: 168k]
  ------------------
  249|   313k|                assert(layout != DAV1D_PIXEL_LAYOUT_I400);
  ------------------
  |  Branch (249:17): [True: 168k, False: 0]
  ------------------
  250|       |
  251|   168k|                const int uvdir = uv_pri_lvl ? uv_dir[dir] : 0;
  ------------------
  |  Branch (251:35): [True: 161k, False: 6.29k]
  ------------------
  252|   504k|                for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (252:34): [True: 336k, False: 168k]
  ------------------
  253|   336k|                    if (!have_tt) goto st_uv;
  ------------------
  |  Branch (253:25): [True: 336k, False: 0]
  ------------------
  254|      0|                    if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (254:25): [True: 0, False: 0]
  |  Branch (254:40): [True: 0, False: 0]
  ------------------
  255|      0|                        if (resize) {
  ------------------
  |  Branch (255:29): [True: 0, False: 0]
  ------------------
  256|      0|                            offset = (sby - 1) * 4 * uv_stride + (bx * 4 >> ss_hor);
  257|      0|                            top = &f->lf.cdef_lpf_line[pl][offset];
  258|      0|                        } else {
  259|      0|                            const int line = sby * (4 << sb128) - 4;
  260|      0|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  261|      0|                            top = &f->lf.lr_lpf_line[pl][offset];
  262|      0|                        }
  263|      0|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  264|      0|                    } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (264:32): [True: 0, False: 0]
  |  Branch (264:48): [True: 0, False: 0]
  ------------------
  265|      0|                        const ptrdiff_t top_offset = sby * 8 * uv_stride +
  266|      0|                                                     (bx * 4 >> ss_hor);
  267|      0|                        top = &f->lf.cdef_line[tf][pl][top_offset];
  268|      0|                        if (resize) {
  ------------------
  |  Branch (268:29): [True: 0, False: 0]
  ------------------
  269|      0|                            offset = (sby * 4 + 2) * uv_stride + (bx * 4 >> ss_hor);
  270|      0|                            bot = &f->lf.cdef_lpf_line[pl][offset];
  271|      0|                        } else {
  272|      0|                            const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  273|      0|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  274|      0|                            bot = &f->lf.lr_lpf_line[pl][offset];
  275|      0|                        }
  276|      0|                    } else {
  277|   336k|                st_uv:;
  278|   336k|                        const ptrdiff_t offset = sby * 8 * uv_stride;
  279|   336k|                        top = &f->lf.cdef_line[tf][pl][have_tt * offset + (bx * 4 >> ss_hor)];
  280|   336k|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  281|   336k|                    }
  282|   336k|                    dsp->cdef.fb[uv_idx](bptrs[pl], f->cur.stride[1],
  283|   336k|                                         lr_bak[bit][pl], top, bot,
  284|   336k|                                         uv_pri_lvl, uv_sec_lvl, uvdir,
  285|   336k|                                         damping - 1, edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   336k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  286|   336k|                }
  287|       |
  288|   313k|            skip_uv:
  289|   313k|                bit ^= 1;
  290|       |
  291|   328k|            next_b:
  292|   328k|                bptrs[0] += 8;
  293|   328k|                bptrs[1] += 8 >> ss_hor;
  294|   328k|                bptrs[2] += 8 >> ss_hor;
  295|   328k|            }
  296|       |
  297|   681k|        next_sb:
  298|   681k|            iptrs[0] += sbsz * 4;
  299|   681k|            iptrs[1] += sbsz * 4 >> ss_hor;
  300|   681k|            iptrs[2] += sbsz * 4 >> ss_hor;
  301|   681k|        }
  302|       |
  303|   287k|        ptrs[0] += 8 * PXSTRIDE(f->cur.stride[0]);
  304|   287k|        ptrs[1] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  305|   287k|        ptrs[2] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  306|   287k|        tc->top_pre_cdef_toggle ^= 1;
  307|   287k|    }
  308|  47.6k|}

dav1d_cdef_dsp_init_8bpc:
  321|  3.66k|COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) {
  322|  3.66k|    c->dir = cdef_find_dir_c;
  323|  3.66k|    c->fb[0] = cdef_filter_block_8x8_c;
  324|  3.66k|    c->fb[1] = cdef_filter_block_4x8_c;
  325|  3.66k|    c->fb[2] = cdef_filter_block_4x4_c;
  326|       |
  327|  3.66k|#if HAVE_ASM
  328|       |#if ARCH_AARCH64 || ARCH_ARM
  329|       |    cdef_dsp_init_arm(c);
  330|       |#elif ARCH_PPC64LE
  331|       |    cdef_dsp_init_ppc(c);
  332|       |#elif ARCH_RISCV
  333|       |    cdef_dsp_init_riscv(c);
  334|       |#elif ARCH_X86
  335|       |    cdef_dsp_init_x86(c);
  336|       |#elif ARCH_LOONGARCH64
  337|       |    cdef_dsp_init_loongarch(c);
  338|       |#endif
  339|  3.66k|#endif
  340|  3.66k|}
dav1d_cdef_dsp_init_16bpc:
  321|  4.99k|COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) {
  322|  4.99k|    c->dir = cdef_find_dir_c;
  323|  4.99k|    c->fb[0] = cdef_filter_block_8x8_c;
  324|  4.99k|    c->fb[1] = cdef_filter_block_4x8_c;
  325|  4.99k|    c->fb[2] = cdef_filter_block_4x4_c;
  326|       |
  327|  4.99k|#if HAVE_ASM
  328|       |#if ARCH_AARCH64 || ARCH_ARM
  329|       |    cdef_dsp_init_arm(c);
  330|       |#elif ARCH_PPC64LE
  331|       |    cdef_dsp_init_ppc(c);
  332|       |#elif ARCH_RISCV
  333|       |    cdef_dsp_init_riscv(c);
  334|       |#elif ARCH_X86
  335|       |    cdef_dsp_init_x86(c);
  336|       |#elif ARCH_LOONGARCH64
  337|       |    cdef_dsp_init_loongarch(c);
  338|       |#endif
  339|  4.99k|#endif
  340|  4.99k|}

dav1d_cdf_thread_update:
 3918|  13.4k|{
 3919|  13.4k|#define update_cdf_1d(n1d, name) \
 3920|  13.4k|    do { \
 3921|  13.4k|        dst->name[n1d] = 0; \
 3922|  13.4k|    } while (0)
 3923|  13.4k|#define update_cdf_2d(n1d, n2d, name) \
 3924|  13.4k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
 3925|  13.4k|#define update_cdf_3d(n1d, n2d, n3d, name) \
 3926|  13.4k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
 3927|  13.4k|#define update_cdf_4d(n1d, n2d, n3d, n4d, name) \
 3928|  13.4k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
 3929|       |
 3930|  13.4k|    memcpy(dst, src, offsetof(CdfContext, m.intrabc));
 3931|       |
 3932|  13.4k|    update_cdf_3d(2, 2, 4, coef.eob_bin_16);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 53.7k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3933|  13.4k|    update_cdf_3d(2, 2, 5, coef.eob_bin_32);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 53.7k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3934|  13.4k|    update_cdf_3d(2, 2, 6, coef.eob_bin_64);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 53.7k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3935|  13.4k|    update_cdf_3d(2, 2, 7, coef.eob_bin_128);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 53.7k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3936|  13.4k|    update_cdf_3d(2, 2, 8, coef.eob_bin_256);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 53.7k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3937|  13.4k|    update_cdf_2d(2, 9, coef.eob_bin_512);
  ------------------
  |  | 3924|  40.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  26.8k|    do { \
  |  |  |  | 3921|  26.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  26.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3938|  13.4k|    update_cdf_2d(2, 10, coef.eob_bin_1024);
  ------------------
  |  | 3924|  40.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  26.8k|    do { \
  |  |  |  | 3921|  26.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  26.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3939|  13.4k|    update_cdf_4d(N_TX_SIZES, 2, 4, 2, coef.eob_base_tok);
  ------------------
  |  | 3928|  80.5k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   201k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|   671k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|   537k|    do { \
  |  |  |  |  |  |  |  | 3921|   537k|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|   537k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 537k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 537k, False: 134k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 134k, False: 67.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 67.1k, False: 13.4k]
  |  |  ------------------
  ------------------
 3940|  13.4k|    update_cdf_4d(N_TX_SIZES, 2, 41 /*42*/, 3, coef.base_tok);
  ------------------
  |  | 3928|  80.5k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   201k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  5.64M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|  5.50M|    do { \
  |  |  |  |  |  |  |  | 3921|  5.50M|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|  5.50M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 5.50M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 5.50M, False: 134k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 134k, False: 67.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 67.1k, False: 13.4k]
  |  |  ------------------
  ------------------
 3941|  13.4k|    update_cdf_4d(4, 2, 21, 3, coef.br_tok);
  ------------------
  |  | 3928|  67.1k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   161k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  2.36M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|  2.25M|    do { \
  |  |  |  |  |  |  |  | 3921|  2.25M|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|  2.25M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 2.25M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 2.25M, False: 107k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 107k, False: 53.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 53.7k, False: 13.4k]
  |  |  ------------------
  ------------------
 3942|  13.4k|    update_cdf_4d(N_TX_SIZES, 2, 9, 1, coef.eob_hi_bit);
  ------------------
  |  | 3928|  80.5k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   201k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  1.34M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|  1.20M|    do { \
  |  |  |  |  |  |  |  | 3921|  1.20M|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|  1.20M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 1.20M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 1.20M, False: 134k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 134k, False: 67.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 67.1k, False: 13.4k]
  |  |  ------------------
  ------------------
 3943|  13.4k|    update_cdf_3d(N_TX_SIZES, 13, 1, coef.skip);
  ------------------
  |  | 3926|  80.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   940k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   873k|    do { \
  |  |  |  |  |  | 3921|   873k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   873k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 873k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 873k, False: 67.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 67.1k, False: 13.4k]
  |  |  ------------------
  ------------------
 3944|  13.4k|    update_cdf_3d(2, 3, 1, coef.dc_sign);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   107k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  80.5k|    do { \
  |  |  |  |  |  | 3921|  80.5k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  80.5k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 80.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 80.5k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3945|       |
 3946|  13.4k|    update_cdf_3d(2, N_INTRA_PRED_MODES, N_UV_INTRA_PRED_MODES - 1 - !k, m.uv_mode);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   376k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   349k|    do { \
  |  |  |  |  |  | 3921|   349k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   349k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 349k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 349k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3947|  13.4k|    update_cdf_2d(4, N_PARTITIONS - 3, m.partition[BL_128X128]);
  ------------------
  |  | 3924|  67.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 53.7k, False: 13.4k]
  |  |  ------------------
  ------------------
 3948|  53.7k|    for (int k = BL_64X64; k < BL_8X8; k++)
  ------------------
  |  Branch (3948:28): [True: 40.2k, False: 13.4k]
  ------------------
 3949|  40.2k|        update_cdf_2d(4, N_PARTITIONS - 1, m.partition[k]);
  ------------------
  |  | 3924|   201k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   161k|    do { \
  |  |  |  | 3921|   161k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   161k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 161k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 161k, False: 40.2k]
  |  |  ------------------
  ------------------
 3950|  13.4k|    update_cdf_2d(4, N_SUB8X8_PARTITIONS - 1, m.partition[BL_8X8]);
  ------------------
  |  | 3924|  67.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 53.7k, False: 13.4k]
  |  |  ------------------
  ------------------
 3951|  13.4k|    update_cdf_2d(6, 15, m.cfl_alpha);
  ------------------
  |  | 3924|  94.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  80.5k|    do { \
  |  |  |  | 3921|  80.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  80.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 80.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 80.5k, False: 13.4k]
  |  |  ------------------
  ------------------
 3952|  13.4k|    update_cdf_2d(2, 15, m.txtp_inter1);
  ------------------
  |  | 3924|  40.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  26.8k|    do { \
  |  |  |  | 3921|  26.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  26.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3953|  13.4k|    update_cdf_1d(11, m.txtp_inter2);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3954|  13.4k|    update_cdf_3d(2, N_INTRA_PRED_MODES, 6, m.txtp_intra1);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   376k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   349k|    do { \
  |  |  |  |  |  | 3921|   349k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   349k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 349k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 349k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3955|  13.4k|    update_cdf_3d(3, N_INTRA_PRED_MODES, 4, m.txtp_intra2);
  ------------------
  |  | 3926|  53.7k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   564k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   523k|    do { \
  |  |  |  |  |  | 3921|   523k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   523k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 523k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 523k, False: 40.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 40.2k, False: 13.4k]
  |  |  ------------------
  ------------------
 3956|  13.4k|    update_cdf_1d(7, m.cfl_sign);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3957|  13.4k|    update_cdf_2d(8, 6, m.angle_delta);
  ------------------
  |  | 3924|   120k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   107k|    do { \
  |  |  |  | 3921|   107k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   107k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 107k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 107k, False: 13.4k]
  |  |  ------------------
  ------------------
 3958|  13.4k|    update_cdf_1d(4, m.filter_intra);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3959|  13.4k|    update_cdf_2d(3, DAV1D_MAX_SEGMENTS - 1, m.seg_id);
  ------------------
  |  | 3924|  53.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  40.2k|    do { \
  |  |  |  | 3921|  40.2k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  40.2k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 40.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 40.2k, False: 13.4k]
  |  |  ------------------
  ------------------
 3960|  13.4k|    update_cdf_3d(2, 7, 6, m.pal_sz);
  ------------------
  |  | 3926|  40.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   214k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   188k|    do { \
  |  |  |  |  |  | 3921|   188k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   188k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 188k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 188k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3961|  13.4k|    update_cdf_4d(2, 7, 5, k + 1, m.color_map);
  ------------------
  |  | 3928|  40.2k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   214k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  1.12M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|   940k|    do { \
  |  |  |  |  |  |  |  | 3921|   940k|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|   940k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 940k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 940k, False: 188k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 188k, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3962|  13.4k|    update_cdf_3d(N_TX_SIZES - 1, 3, imin(k + 1, 2), m.txsz);
  ------------------
  |  | 3926|  67.1k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   214k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   161k|    do { \
  |  |  |  |  |  | 3921|   161k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   161k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 161k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 161k, False: 53.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 53.7k, False: 13.4k]
  |  |  ------------------
  ------------------
 3963|  13.4k|    update_cdf_1d(3, m.delta_q);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3964|  13.4k|    update_cdf_2d(5, 3, m.delta_lf);
  ------------------
  |  | 3924|  80.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  67.1k|    do { \
  |  |  |  | 3921|  67.1k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  67.1k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 67.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 67.1k, False: 13.4k]
  |  |  ------------------
  ------------------
 3965|  13.4k|    update_cdf_1d(2, m.restore_switchable);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3966|  13.4k|    update_cdf_1d(1, m.restore_wiener);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3967|  13.4k|    update_cdf_1d(1, m.restore_sgrproj);
  ------------------
  |  | 3920|  13.4k|    do { \
  |  | 3921|  13.4k|        dst->name[n1d] = 0; \
  |  | 3922|  13.4k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 13.4k]
  |  |  ------------------
  ------------------
 3968|  13.4k|    update_cdf_2d(4, 1, m.txtp_inter3);
  ------------------
  |  | 3924|  67.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.7k|    do { \
  |  |  |  | 3921|  53.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  53.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 53.7k, False: 13.4k]
  |  |  ------------------
  ------------------
 3969|  13.4k|    update_cdf_2d(N_BS_SIZES, 1, m.use_filter_intra);
  ------------------
  |  | 3924|   308k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   295k|    do { \
  |  |  |  | 3921|   295k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   295k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 295k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 295k, False: 13.4k]
  |  |  ------------------
  ------------------
 3970|  13.4k|    update_cdf_3d(7, 3, 1, m.txpart);
  ------------------
  |  | 3926|   107k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   376k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   282k|    do { \
  |  |  |  |  |  | 3921|   282k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   282k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 282k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 282k, False: 94.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 94.0k, False: 13.4k]
  |  |  ------------------
  ------------------
 3971|  13.4k|    update_cdf_2d(3, 1, m.skip);
  ------------------
  |  | 3924|  53.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  40.2k|    do { \
  |  |  |  | 3921|  40.2k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  40.2k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 40.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 40.2k, False: 13.4k]
  |  |  ------------------
  ------------------
 3972|  13.4k|    update_cdf_3d(7, 3, 1, m.pal_y);
  ------------------
  |  | 3926|   107k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   376k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   282k|    do { \
  |  |  |  |  |  | 3921|   282k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   282k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 282k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 282k, False: 94.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 94.0k, False: 13.4k]
  |  |  ------------------
  ------------------
 3973|  13.4k|    update_cdf_2d(2, 1, m.pal_uv);
  ------------------
  |  | 3924|  40.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  26.8k|    do { \
  |  |  |  | 3921|  26.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  26.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 26.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 26.8k, False: 13.4k]
  |  |  ------------------
  ------------------
 3974|       |
 3975|  13.4k|    if (IS_KEY_OR_INTRA(hdr))
  ------------------
  |  |   43|  13.4k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  13.4k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 5.83k, False: 7.59k]
  |  |  ------------------
  ------------------
 3976|  5.83k|        return;
 3977|       |
 3978|  7.59k|    memcpy(dst->m.y_mode, src->m.y_mode,
 3979|  7.59k|           offsetof(CdfContext, kfym) - offsetof(CdfContext, m.y_mode));
 3980|       |
 3981|  7.59k|    update_cdf_2d(4, N_INTRA_PRED_MODES - 1, m.y_mode);
  ------------------
  |  | 3924|  37.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  37.9k|    do { \
  |  |  |  | 3921|  30.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  30.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 30.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 30.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 3982|  7.59k|    update_cdf_2d(9, 15, m.wedge_idx);
  ------------------
  |  | 3924|  75.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  75.9k|    do { \
  |  |  |  | 3921|  68.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  68.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 68.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 68.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 3983|  7.59k|    update_cdf_2d(8, N_COMP_INTER_PRED_MODES - 1, m.comp_inter_mode);
  ------------------
  |  | 3924|  68.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  68.3k|    do { \
  |  |  |  | 3921|  60.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  60.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 60.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 60.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 3984|  7.59k|    update_cdf_3d(2, 8, DAV1D_N_SWITCHABLE_FILTERS - 1, m.filter);
  ------------------
  |  | 3926|  22.7k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   136k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   129k|    do { \
  |  |  |  |  |  | 3921|   121k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   121k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 121k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 121k, False: 15.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 15.1k, False: 7.59k]
  |  |  ------------------
  ------------------
 3985|  7.59k|    update_cdf_2d(4, 3, m.interintra_mode);
  ------------------
  |  | 3924|  37.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  37.9k|    do { \
  |  |  |  | 3921|  30.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  30.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 30.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 30.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 3986|  7.59k|    update_cdf_2d(N_BS_SIZES, 2, m.motion_mode);
  ------------------
  |  | 3924|   174k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   174k|    do { \
  |  |  |  | 3921|   167k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   167k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 167k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 167k, False: 7.59k]
  |  |  ------------------
  ------------------
 3987|  7.59k|    update_cdf_2d(3, 1, m.skip_mode);
  ------------------
  |  | 3924|  30.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  30.3k|    do { \
  |  |  |  | 3921|  22.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  22.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 22.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 22.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 3988|  7.59k|    update_cdf_2d(6, 1, m.newmv_mode);
  ------------------
  |  | 3924|  53.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.1k|    do { \
  |  |  |  | 3921|  45.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  45.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 45.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 45.5k, False: 7.59k]
  |  |  ------------------
  ------------------
 3989|  7.59k|    update_cdf_2d(2, 1, m.globalmv_mode);
  ------------------
  |  | 3924|  22.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  22.7k|    do { \
  |  |  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  15.1k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 15.1k, False: 7.59k]
  |  |  ------------------
  ------------------
 3990|  7.59k|    update_cdf_2d(6, 1, m.refmv_mode);
  ------------------
  |  | 3924|  53.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.1k|    do { \
  |  |  |  | 3921|  45.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  45.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 45.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 45.5k, False: 7.59k]
  |  |  ------------------
  ------------------
 3991|  7.59k|    update_cdf_2d(3, 1, m.drl_bit);
  ------------------
  |  | 3924|  30.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  30.3k|    do { \
  |  |  |  | 3921|  22.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  22.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 22.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 22.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 3992|  7.59k|    update_cdf_2d(4, 1, m.intra);
  ------------------
  |  | 3924|  37.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  37.9k|    do { \
  |  |  |  | 3921|  30.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  30.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 30.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 30.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 3993|  7.59k|    update_cdf_2d(5, 1, m.comp);
  ------------------
  |  | 3924|  45.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  45.5k|    do { \
  |  |  |  | 3921|  37.9k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  37.9k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 37.9k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 37.9k, False: 7.59k]
  |  |  ------------------
  ------------------
 3994|  7.59k|    update_cdf_2d(5, 1, m.comp_dir);
  ------------------
  |  | 3924|  45.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  45.5k|    do { \
  |  |  |  | 3921|  37.9k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  37.9k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 37.9k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 37.9k, False: 7.59k]
  |  |  ------------------
  ------------------
 3995|  7.59k|    update_cdf_2d(6, 1, m.jnt_comp);
  ------------------
  |  | 3924|  53.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.1k|    do { \
  |  |  |  | 3921|  45.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  45.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 45.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 45.5k, False: 7.59k]
  |  |  ------------------
  ------------------
 3996|  7.59k|    update_cdf_2d(6, 1, m.mask_comp);
  ------------------
  |  | 3924|  53.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  53.1k|    do { \
  |  |  |  | 3921|  45.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  45.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 45.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 45.5k, False: 7.59k]
  |  |  ------------------
  ------------------
 3997|  7.59k|    update_cdf_2d(9, 1, m.wedge_comp);
  ------------------
  |  | 3924|  75.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  75.9k|    do { \
  |  |  |  | 3921|  68.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  68.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 68.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 68.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 3998|  7.59k|    update_cdf_3d(6, 3, 1, m.ref);
  ------------------
  |  | 3926|  53.1k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   182k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   144k|    do { \
  |  |  |  |  |  | 3921|   136k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   136k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 136k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 136k, False: 45.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 45.5k, False: 7.59k]
  |  |  ------------------
  ------------------
 3999|  7.59k|    update_cdf_3d(3, 3, 1, m.comp_fwd_ref);
  ------------------
  |  | 3926|  30.3k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  91.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  75.9k|    do { \
  |  |  |  |  |  | 3921|  68.3k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  68.3k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 68.3k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 68.3k, False: 22.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 22.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 4000|  7.59k|    update_cdf_3d(2, 3, 1, m.comp_bwd_ref);
  ------------------
  |  | 3926|  22.7k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  60.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  53.1k|    do { \
  |  |  |  |  |  | 3921|  45.5k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  45.5k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 45.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 45.5k, False: 15.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 15.1k, False: 7.59k]
  |  |  ------------------
  ------------------
 4001|  7.59k|    update_cdf_3d(3, 3, 1, m.comp_uni_ref);
  ------------------
  |  | 3926|  30.3k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  91.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  75.9k|    do { \
  |  |  |  |  |  | 3921|  68.3k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  68.3k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 68.3k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 68.3k, False: 22.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 22.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 4002|  7.59k|    update_cdf_2d(3, 1, m.seg_pred);
  ------------------
  |  | 3924|  30.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  30.3k|    do { \
  |  |  |  | 3921|  22.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  22.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 22.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 22.7k, False: 7.59k]
  |  |  ------------------
  ------------------
 4003|  7.59k|    update_cdf_2d(4, 1, m.interintra);
  ------------------
  |  | 3924|  37.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  37.9k|    do { \
  |  |  |  | 3921|  30.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  30.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 30.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 30.3k, False: 7.59k]
  |  |  ------------------
  ------------------
 4004|  7.59k|    update_cdf_2d(7, 1, m.interintra_wedge);
  ------------------
  |  | 3924|  60.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  60.7k|    do { \
  |  |  |  | 3921|  53.1k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  53.1k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 53.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 53.1k, False: 7.59k]
  |  |  ------------------
  ------------------
 4005|  7.59k|    update_cdf_2d(N_BS_SIZES, 1, m.obmc);
  ------------------
  |  | 3924|   174k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   174k|    do { \
  |  |  |  | 3921|   167k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   167k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 167k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 167k, False: 7.59k]
  |  |  ------------------
  ------------------
 4006|       |
 4007|  22.7k|    for (int k = 0; k < 2; k++) {
  ------------------
  |  Branch (4007:21): [True: 15.1k, False: 7.59k]
  ------------------
 4008|  15.1k|        update_cdf_1d(10, mv.comp[k].classes);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4009|  15.1k|        update_cdf_1d(1, mv.comp[k].sign);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4010|  15.1k|        update_cdf_1d(1, mv.comp[k].class0);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4011|  15.1k|        update_cdf_2d(2, 3, mv.comp[k].class0_fp);
  ------------------
  |  | 3924|  45.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  30.3k|    do { \
  |  |  |  | 3921|  30.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  30.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 30.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 30.3k, False: 15.1k]
  |  |  ------------------
  ------------------
 4012|  15.1k|        update_cdf_1d(1, mv.comp[k].class0_hp);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4013|  15.1k|        update_cdf_2d(10, 1, mv.comp[k].classN);
  ------------------
  |  | 3924|   167k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   151k|    do { \
  |  |  |  | 3921|   151k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   151k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 151k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 151k, False: 15.1k]
  |  |  ------------------
  ------------------
 4014|  15.1k|        update_cdf_1d(3, mv.comp[k].classN_fp);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4015|  15.1k|        update_cdf_1d(1, mv.comp[k].classN_hp);
  ------------------
  |  | 3920|  15.1k|    do { \
  |  | 3921|  15.1k|        dst->name[n1d] = 0; \
  |  | 3922|  15.1k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 15.1k]
  |  |  ------------------
  ------------------
 4016|  15.1k|    }
 4017|  7.59k|    update_cdf_1d(N_MV_JOINTS - 1, mv.joint);
  ------------------
  |  | 3920|  7.59k|    do { \
  |  | 3921|  7.59k|        dst->name[n1d] = 0; \
  |  | 3922|  7.59k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 7.59k]
  |  |  ------------------
  ------------------
 4018|  7.59k|}
dav1d_cdf_thread_init_static:
 4023|  33.7k|void dav1d_cdf_thread_init_static(CdfThreadContext *const cdf, const unsigned qidx) {
 4024|       |    cdf->ref = NULL;
 4025|  33.7k|    cdf->data.qcat = (qidx > 20) + (qidx > 60) + (qidx > 120);
 4026|  33.7k|}
dav1d_cdf_thread_copy:
 4028|  66.2k|void dav1d_cdf_thread_copy(CdfContext *const dst, const CdfThreadContext *const src) {
 4029|  66.2k|    if (src->ref) {
  ------------------
  |  Branch (4029:9): [True: 16.6k, False: 49.5k]
  ------------------
 4030|  16.6k|        memcpy(dst, src->data.cdf, sizeof(*dst));
 4031|  49.5k|    } else {
 4032|  49.5k|        dst->coef = default_coef_cdf[src->data.qcat];
 4033|  49.5k|        memcpy(&dst->m, &default_cdf.m,
 4034|  49.5k|               offsetof(CdfDefaultContext, mv.joint));
 4035|  49.5k|        memcpy(&dst->mv.comp[1], &default_cdf.mv.comp,
 4036|       |               sizeof(default_cdf) - offsetof(CdfDefaultContext, mv.comp));
 4037|  49.5k|    }
 4038|  66.2k|}
dav1d_cdf_thread_alloc:
 4042|  15.7k|{
 4043|  15.7k|    cdf->ref = dav1d_ref_create_using_pool(c->cdf_pool,
 4044|  15.7k|                                           sizeof(CdfContext) + sizeof(atomic_uint));
 4045|  15.7k|    if (!cdf->ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (4045:9): [True: 0, False: 15.7k]
  ------------------
 4046|  15.7k|    cdf->data.cdf = cdf->ref->data;
 4047|  15.7k|    if (have_frame_mt) {
  ------------------
  |  Branch (4047:9): [True: 0, False: 15.7k]
  ------------------
 4048|      0|        cdf->progress = (atomic_uint *) &cdf->data.cdf[1];
 4049|       |        atomic_init(cdf->progress, 0);
 4050|      0|    }
 4051|  15.7k|    return 0;
 4052|  15.7k|}
dav1d_cdf_thread_ref:
 4056|   341k|{
 4057|   341k|    *dst = *src;
 4058|   341k|    if (src->ref)
  ------------------
  |  Branch (4058:9): [True: 94.1k, False: 247k]
  ------------------
 4059|  94.1k|        dav1d_ref_inc(src->ref);
 4060|   341k|}
dav1d_cdf_thread_unref:
 4062|   812k|void dav1d_cdf_thread_unref(CdfThreadContext *const cdf) {
 4063|       |    memset(&cdf->data, 0, sizeof(*cdf) - offsetof(CdfThreadContext, data));
 4064|   812k|    dav1d_ref_dec(&cdf->ref);
 4065|   812k|}

dav1d_init_cpu:
   63|      1|COLD void dav1d_init_cpu(void) {
   64|      1|#if HAVE_ASM && !__has_feature(memory_sanitizer)
   65|       |// memory sanitizer is inherently incompatible with asm
   66|       |#if ARCH_AARCH64 || ARCH_ARM
   67|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_arm();
   68|       |#elif ARCH_LOONGARCH
   69|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_loongarch();
   70|       |#elif ARCH_PPC64LE
   71|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_ppc();
   72|       |#elif ARCH_RISCV
   73|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_riscv();
   74|       |#elif ARCH_X86
   75|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_x86();
   76|      1|#endif
   77|      1|#endif
   78|      1|}

cpu.c:dav1d_get_default_cpu_flags:
   58|      1|static ALWAYS_INLINE unsigned dav1d_get_default_cpu_flags(void) {
   59|      1|    unsigned flags = 0;
   60|       |
   61|       |#if ARCH_AARCH64 || ARCH_ARM
   62|       |#if defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32) || ARCH_AARCH64
   63|       |    flags |= DAV1D_ARM_CPU_FLAG_NEON;
   64|       |#endif
   65|       |#ifdef __ARM_FEATURE_DOTPROD
   66|       |    flags |= DAV1D_ARM_CPU_FLAG_DOTPROD;
   67|       |#endif
   68|       |#ifdef __ARM_FEATURE_MATMUL_INT8
   69|       |    flags |= DAV1D_ARM_CPU_FLAG_I8MM;
   70|       |#endif
   71|       |#if ARCH_AARCH64
   72|       |#ifdef __ARM_FEATURE_SVE
   73|       |    flags |= DAV1D_ARM_CPU_FLAG_SVE;
   74|       |#endif
   75|       |#ifdef __ARM_FEATURE_SVE2
   76|       |    flags |= DAV1D_ARM_CPU_FLAG_SVE2;
   77|       |#endif
   78|       |#endif /* ARCH_AARCH64 */
   79|       |#elif ARCH_PPC64LE
   80|       |#if defined(__VSX__)
   81|       |    flags |= DAV1D_PPC_CPU_FLAG_VSX;
   82|       |#endif
   83|       |#if defined(__POWER9_VECTOR__)
   84|       |    flags |= DAV1D_PPC_CPU_FLAG_PWR9;
   85|       |#endif
   86|       |#elif ARCH_RISCV
   87|       |#if defined(__riscv_v)
   88|       |    flags |= DAV1D_RISCV_CPU_FLAG_V;
   89|       |#endif
   90|       |#elif ARCH_X86
   91|       |#if defined(__AVX512F__) && defined(__AVX512CD__) && \
   92|       |    defined(__AVX512BW__) && defined(__AVX512DQ__) && \
   93|       |    defined(__AVX512VL__) && defined(__AVX512VNNI__) && \
   94|       |    defined(__AVX512IFMA__) && defined(__AVX512VBMI__) && \
   95|       |    defined(__AVX512VBMI2__) && defined(__AVX512VPOPCNTDQ__) && \
   96|       |    defined(__AVX512BITALG__) && defined(__GFNI__) && \
   97|       |    defined(__VAES__) && defined(__VPCLMULQDQ__)
   98|       |    flags |= DAV1D_X86_CPU_FLAG_AVX512ICL |
   99|       |             DAV1D_X86_CPU_FLAG_AVX2 |
  100|       |             DAV1D_X86_CPU_FLAG_SSE41 |
  101|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  102|       |             DAV1D_X86_CPU_FLAG_SSE2;
  103|       |#elif defined(__AVX2__)
  104|       |    flags |= DAV1D_X86_CPU_FLAG_AVX2 |
  105|       |             DAV1D_X86_CPU_FLAG_SSE41 |
  106|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  107|       |             DAV1D_X86_CPU_FLAG_SSE2;
  108|       |#elif defined(__SSE4_1__) || defined(__AVX__)
  109|       |    flags |= DAV1D_X86_CPU_FLAG_SSE41 |
  110|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  111|       |             DAV1D_X86_CPU_FLAG_SSE2;
  112|       |#elif defined(__SSSE3__)
  113|       |    flags |= DAV1D_X86_CPU_FLAG_SSSE3 |
  114|       |             DAV1D_X86_CPU_FLAG_SSE2;
  115|       |#elif ARCH_X86_64 || defined(__SSE2__) || \
  116|       |      (defined(_M_IX86_FP) && _M_IX86_FP >= 2)
  117|       |    flags |= DAV1D_X86_CPU_FLAG_SSE2;
  118|      1|#endif
  119|      1|#endif
  120|       |
  121|      1|    return flags;
  122|      1|}
pal.c:dav1d_get_cpu_flags:
  124|  10.2k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  10.2k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  10.2k|    return flags;
  134|  10.2k|}
refmvs.c:dav1d_get_cpu_flags:
  124|  10.2k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  10.2k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  10.2k|    return flags;
  134|  10.2k|}
msac.c:dav1d_get_cpu_flags:
  124|  50.5k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  50.5k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  50.5k|    return flags;
  134|  50.5k|}
cdef_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
filmgrain_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
ipred_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
itx_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
loopfilter_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
looprestoration_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}
mc_tmpl.c:dav1d_get_cpu_flags:
  124|  8.66k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.66k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.66k|    return flags;
  134|  8.66k|}

ctx.c:memset_w1:
   34|  19.5M|static void memset_w1(void *const ptr, const int value) {
   35|  19.5M|    set_ctx1((uint8_t *) ptr, 0, value);
  ------------------
  |  |   56|  19.5M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  ------------------
   36|  19.5M|}
ctx.c:memset_w2:
   38|  9.05M|static void memset_w2(void *const ptr, const int value) {
   39|  9.05M|    set_ctx2((uint8_t *) ptr, 0, value);
  ------------------
  |  |   58|  9.05M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  ------------------
   40|  9.05M|}
ctx.c:memset_w4:
   42|  7.44M|static void memset_w4(void *const ptr, const int value) {
   43|  7.44M|    set_ctx4((uint8_t *) ptr, 0, value);
  ------------------
  |  |   60|  7.44M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  ------------------
   44|  7.44M|}
ctx.c:memset_w8:
   46|  5.65M|static void memset_w8(void *const ptr, const int value) {
   47|  5.65M|    set_ctx8((uint8_t *) ptr, 0, value);
  ------------------
  |  |   62|  5.65M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  ------------------
   48|  5.65M|}
ctx.c:memset_w16:
   50|  2.34M|static void memset_w16(void *const ptr, const int value) {
   51|  2.34M|    set_ctx16((uint8_t *) ptr, 0, value);
  ------------------
  |  |   63|  2.34M|#define set_ctx16(var, off, val) do { \
  |  |   64|  2.34M|        memset(&(var)[off], val, 16); \
  |  |   65|  2.34M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (65:14): [Folded, False: 2.34M]
  |  |  ------------------
  ------------------
   52|  2.34M|}
ctx.c:memset_w32:
   54|   217k|static void memset_w32(void *const ptr, const int value) {
   55|   217k|    set_ctx32((uint8_t *) ptr, 0, value);
  ------------------
  |  |   66|   217k|#define set_ctx32(var, off, val) do { \
  |  |   67|   217k|        memset(&(var)[off], val, 32); \
  |  |   68|   217k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (68:14): [Folded, False: 217k]
  |  |  ------------------
  ------------------
   56|   217k|}

lf_mask.c:dav1d_memset_likely_pow2:
   44|  2.14M|static inline void dav1d_memset_likely_pow2(void *const ptr, const int value, const int n) {
   45|  2.14M|    assert(n >= 1 && n <= 32);
  ------------------
  |  Branch (45:5): [True: 2.14M, False: 0]
  |  Branch (45:5): [True: 2.14M, False: 0]
  ------------------
   46|  2.14M|    if ((n&(n-1)) == 0) {
  ------------------
  |  Branch (46:9): [True: 2.03M, False: 106k]
  ------------------
   47|  2.03M|        dav1d_memset_pow2[ulog2(n)](ptr, value);
   48|  2.03M|    } else {
   49|   106k|        memset(ptr, value, n);
   50|   106k|    }
   51|  2.14M|}
recon_tmpl.c:dav1d_memset_likely_pow2:
   44|  16.6M|static inline void dav1d_memset_likely_pow2(void *const ptr, const int value, const int n) {
   45|  16.6M|    assert(n >= 1 && n <= 32);
  ------------------
  |  Branch (45:5): [True: 16.6M, False: 0]
  |  Branch (45:5): [True: 16.6M, False: 0]
  ------------------
   46|  16.6M|    if ((n&(n-1)) == 0) {
  ------------------
  |  Branch (46:9): [True: 16.4M, False: 184k]
  ------------------
   47|  16.4M|        dav1d_memset_pow2[ulog2(n)](ptr, value);
   48|  16.4M|    } else {
   49|   184k|        memset(ptr, value, n);
   50|   184k|    }
   51|  16.6M|}

dav1d_data_create_internal:
   43|  78.9k|uint8_t *dav1d_data_create_internal(Dav1dData *const buf, const size_t sz) {
   44|  78.9k|    validate_input_or_ret(buf != NULL, NULL);
  ------------------
  |  |   52|  78.9k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 78.9k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
   45|       |
   46|  78.9k|    if (sz > SIZE_MAX / 2) return NULL;
  ------------------
  |  Branch (46:9): [True: 0, False: 78.9k]
  ------------------
   47|  78.9k|    buf->ref = dav1d_ref_create(ALLOC_DAV1DDATA, sz);
  ------------------
  |  |   49|  78.9k|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
   48|  78.9k|    if (!buf->ref) return NULL;
  ------------------
  |  Branch (48:9): [True: 0, False: 78.9k]
  ------------------
   49|  78.9k|    buf->data = buf->ref->const_data;
   50|  78.9k|    buf->sz = sz;
   51|  78.9k|    dav1d_data_props_set_defaults(&buf->m);
   52|  78.9k|    buf->m.size = sz;
   53|       |
   54|  78.9k|    return buf->ref->data;
   55|  78.9k|}
dav1d_data_ref:
   98|   128k|void dav1d_data_ref(Dav1dData *const dst, const Dav1dData *const src) {
   99|   128k|    assert(dst != NULL);
  ------------------
  |  Branch (99:5): [True: 128k, False: 0]
  ------------------
  100|   128k|    assert(dst->data == NULL);
  ------------------
  |  Branch (100:5): [True: 128k, False: 0]
  ------------------
  101|   128k|    assert(src != NULL);
  ------------------
  |  Branch (101:5): [True: 128k, False: 0]
  ------------------
  102|       |
  103|   128k|    if (src->ref) {
  ------------------
  |  Branch (103:9): [True: 128k, False: 0]
  ------------------
  104|   128k|        assert(src->data != NULL);
  ------------------
  |  Branch (104:9): [True: 128k, False: 0]
  ------------------
  105|   128k|        dav1d_ref_inc(src->ref);
  106|   128k|    }
  107|   128k|    if (src->m.user_data.ref) dav1d_ref_inc(src->m.user_data.ref);
  ------------------
  |  Branch (107:9): [True: 0, False: 128k]
  ------------------
  108|   128k|    *dst = *src;
  109|   128k|}
dav1d_data_props_copy:
  113|   114k|{
  114|   114k|    assert(dst != NULL);
  ------------------
  |  Branch (114:5): [True: 114k, False: 0]
  ------------------
  115|   114k|    assert(src != NULL);
  ------------------
  |  Branch (115:5): [True: 114k, False: 0]
  ------------------
  116|       |
  117|   114k|    dav1d_ref_dec(&dst->user_data.ref);
  118|   114k|    *dst = *src;
  119|   114k|    if (dst->user_data.ref) dav1d_ref_inc(dst->user_data.ref);
  ------------------
  |  Branch (119:9): [True: 0, False: 114k]
  ------------------
  120|   114k|}
dav1d_data_props_set_defaults:
  122|  1.04M|void dav1d_data_props_set_defaults(Dav1dDataProps *const props) {
  123|  1.04M|    assert(props != NULL);
  ------------------
  |  Branch (123:5): [True: 1.04M, False: 0]
  ------------------
  124|       |
  125|  1.04M|    memset(props, 0, sizeof(*props));
  126|       |    props->timestamp = INT64_MIN;
  127|  1.04M|    props->offset = -1;
  128|  1.04M|}
dav1d_data_props_unref_internal:
  130|  10.2k|void dav1d_data_props_unref_internal(Dav1dDataProps *const props) {
  131|  10.2k|    validate_input(props != NULL);
  ------------------
  |  |   59|  10.2k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  10.2k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  132|       |
  133|  10.2k|    struct Dav1dRef *user_data_ref = props->user_data.ref;
  134|  10.2k|    dav1d_data_props_set_defaults(props);
  135|  10.2k|    dav1d_ref_dec(&user_data_ref);
  136|  10.2k|}
dav1d_data_unref_internal:
  138|   218k|void dav1d_data_unref_internal(Dav1dData *const buf) {
  139|   218k|    validate_input(buf != NULL);
  ------------------
  |  |   59|   218k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|   218k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 218k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  140|       |
  141|   218k|    struct Dav1dRef *user_data_ref = buf->m.user_data.ref;
  142|   218k|    if (buf->ref) {
  ------------------
  |  Branch (142:9): [True: 207k, False: 10.2k]
  ------------------
  143|   207k|        validate_input(buf->data != NULL);
  ------------------
  |  |   59|   207k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|   207k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 207k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  144|   207k|        dav1d_ref_dec(&buf->ref);
  145|   207k|    }
  146|   218k|    memset(buf, 0, sizeof(*buf));
  147|   218k|    dav1d_data_props_set_defaults(&buf->m);
  148|   218k|    dav1d_ref_dec(&user_data_ref);
  149|   218k|}

dav1d_decode_tile_sbrow:
 2597|   179k|int dav1d_decode_tile_sbrow(Dav1dTaskContext *const t) {
 2598|   179k|    const Dav1dFrameContext *const f = t->f;
 2599|   179k|    const enum BlockLevel root_bl = f->seq_hdr->sb128 ? BL_128X128 : BL_64X64;
  ------------------
  |  Branch (2599:37): [True: 109k, False: 70.1k]
  ------------------
 2600|   179k|    Dav1dTileState *const ts = t->ts;
 2601|   179k|    const Dav1dContext *const c = f->c;
 2602|   179k|    const int sb_step = f->sb_step;
 2603|   179k|    const int tile_row = ts->tiling.row, tile_col = ts->tiling.col;
 2604|   179k|    const int col_sb_start = f->frame_hdr->tiling.col_start_sb[tile_col];
 2605|   179k|    const int col_sb128_start = col_sb_start >> !f->seq_hdr->sb128;
 2606|       |
 2607|   179k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|   358k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 54.1k, False: 125k]
  |  |  ------------------
  ------------------
  |  Branch (2607:45): [True: 87.0k, False: 37.9k]
  ------------------
 2608|   141k|        dav1d_refmvs_tile_sbrow_init(&t->rt, &f->rf, ts->tiling.col_start,
 2609|   141k|                                     ts->tiling.col_end, ts->tiling.row_start,
 2610|   141k|                                     ts->tiling.row_end, t->by >> f->sb_shift,
 2611|   141k|                                     ts->tiling.row, t->frame_thread.pass);
 2612|   141k|    }
 2613|       |
 2614|   179k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) && c->n_fc > 1) {
  ------------------
  |  |   36|   358k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 54.1k, False: 125k]
  |  |  ------------------
  ------------------
  |  Branch (2614:45): [True: 0, False: 54.1k]
  ------------------
 2615|      0|        const int sby = (t->by - ts->tiling.row_start) >> f->sb_shift;
 2616|      0|        int (*const lowest_px)[2] = ts->lowest_pixel[sby];
 2617|      0|        for (int n = 0; n < 7; n++)
  ------------------
  |  Branch (2617:25): [True: 0, False: 0]
  ------------------
 2618|      0|            for (int m = 0; m < 2; m++)
  ------------------
  |  Branch (2618:29): [True: 0, False: 0]
  ------------------
 2619|      0|                lowest_px[n][m] = INT_MIN;
 2620|      0|    }
 2621|       |
 2622|   179k|    reset_context(&t->l, IS_KEY_OR_INTRA(f->frame_hdr), t->frame_thread.pass);
  ------------------
  |  |   43|   179k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|   179k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
 2623|   179k|    if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2623:9): [True: 0, False: 179k]
  ------------------
 2624|      0|        const int off_2pass = c->n_tc > 1 ? f->sb128w * f->frame_hdr->tiling.rows : 0;
  ------------------
  |  Branch (2624:31): [True: 0, False: 0]
  ------------------
 2625|      0|        for (t->bx = ts->tiling.col_start,
 2626|      0|             t->a = f->a + off_2pass + col_sb128_start + tile_row * f->sb128w;
 2627|      0|             t->bx < ts->tiling.col_end; t->bx += sb_step)
  ------------------
  |  Branch (2627:14): [True: 0, False: 0]
  ------------------
 2628|      0|        {
 2629|      0|            if (atomic_load_explicit(c->flush, memory_order_acquire))
  ------------------
  |  Branch (2629:17): [True: 0, False: 0]
  ------------------
 2630|      0|                return 1;
 2631|      0|            if (decode_sb(t, root_bl, dav1d_intra_edge_tree[root_bl]))
  ------------------
  |  Branch (2631:17): [True: 0, False: 0]
  ------------------
 2632|      0|                return 1;
 2633|      0|            if (t->bx & 16 || f->seq_hdr->sb128)
  ------------------
  |  Branch (2633:17): [True: 0, False: 0]
  |  Branch (2633:31): [True: 0, False: 0]
  ------------------
 2634|      0|                t->a++;
 2635|      0|        }
 2636|      0|        f->bd_fn.backup_ipred_edge(t);
 2637|      0|        return 0;
 2638|      0|    }
 2639|       |
 2640|   179k|    if (f->c->n_tc > 1 && f->frame_hdr->use_ref_frame_mvs) {
  ------------------
  |  Branch (2640:9): [True: 0, False: 179k]
  |  Branch (2640:27): [True: 0, False: 0]
  ------------------
 2641|      0|        f->c->refmvs_dsp.load_tmvs(&f->rf, ts->tiling.row,
 2642|      0|                                   ts->tiling.col_start >> 1, ts->tiling.col_end >> 1,
 2643|      0|                                   t->by >> 1, (t->by + sb_step) >> 1);
 2644|      0|    }
 2645|   179k|    memset(t->pal_sz_uv[1], 0, sizeof(*t->pal_sz_uv));
 2646|   179k|    const int sb128y = t->by >> 5;
 2647|   179k|    for (t->bx = ts->tiling.col_start, t->a = f->a + col_sb128_start + tile_row * f->sb128w,
 2648|   179k|         t->lf_mask = f->lf.mask + sb128y * f->sb128w + col_sb128_start;
 2649|   772k|         t->bx < ts->tiling.col_end; t->bx += sb_step)
  ------------------
  |  Branch (2649:10): [True: 601k, False: 170k]
  ------------------
 2650|   601k|    {
 2651|   601k|        if (atomic_load_explicit(c->flush, memory_order_acquire))
  ------------------
  |  Branch (2651:13): [True: 0, False: 601k]
  ------------------
 2652|      0|            return 1;
 2653|   601k|        if (root_bl == BL_128X128) {
  ------------------
  |  Branch (2653:13): [True: 234k, False: 367k]
  ------------------
 2654|   234k|            t->cur_sb_cdef_idx_ptr = t->lf_mask->cdef_idx;
 2655|   234k|            t->cur_sb_cdef_idx_ptr[0] = -1;
 2656|   234k|            t->cur_sb_cdef_idx_ptr[1] = -1;
 2657|   234k|            t->cur_sb_cdef_idx_ptr[2] = -1;
 2658|   234k|            t->cur_sb_cdef_idx_ptr[3] = -1;
 2659|   367k|        } else {
 2660|   367k|            t->cur_sb_cdef_idx_ptr =
 2661|   367k|                &t->lf_mask->cdef_idx[((t->bx & 16) >> 4) +
 2662|   367k|                                      ((t->by & 16) >> 3)];
 2663|   367k|            t->cur_sb_cdef_idx_ptr[0] = -1;
 2664|   367k|        }
 2665|       |        // Restoration filter
 2666|  2.40M|        for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (2666:25): [True: 1.80M, False: 601k]
  ------------------
 2667|  1.80M|            if (!((f->lf.restore_planes >> p) & 1U))
  ------------------
  |  Branch (2667:17): [True: 1.65M, False: 153k]
  ------------------
 2668|  1.65M|                continue;
 2669|       |
 2670|   153k|            const int ss_ver = p && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2670:32): [True: 69.5k, False: 83.8k]
  |  Branch (2670:37): [True: 27.3k, False: 42.2k]
  ------------------
 2671|   153k|            const int ss_hor = p && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2671:32): [True: 69.5k, False: 83.8k]
  |  Branch (2671:37): [True: 31.6k, False: 37.9k]
  ------------------
 2672|   153k|            const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!p];
 2673|   153k|            const int y = t->by * 4 >> ss_ver;
 2674|   153k|            const int h = (f->cur.p.h + ss_ver) >> ss_ver;
 2675|       |
 2676|   153k|            const int unit_size = 1 << unit_size_log2;
 2677|   153k|            const unsigned mask = unit_size - 1;
 2678|   153k|            if (y & mask) continue;
  ------------------
  |  Branch (2678:17): [True: 33.6k, False: 119k]
  ------------------
 2679|   119k|            const int half_unit = unit_size >> 1;
 2680|       |            // Round half up at frame boundaries, if there's more than one
 2681|       |            // restoration unit
 2682|   119k|            if (y && y + half_unit > h) continue;
  ------------------
  |  Branch (2682:17): [True: 32.6k, False: 87.1k]
  |  Branch (2682:22): [True: 2.43k, False: 30.2k]
  ------------------
 2683|       |
 2684|   117k|            const enum Dav1dRestorationType frame_type = f->frame_hdr->restoration.type[p];
 2685|       |
 2686|   117k|            if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2686:17): [True: 21.5k, False: 95.8k]
  ------------------
 2687|  21.5k|                const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2688|  21.5k|                const int n_units = imax(1, (w + half_unit) >> unit_size_log2);
 2689|       |
 2690|  21.5k|                const int d = f->frame_hdr->super_res.width_scale_denominator;
 2691|  21.5k|                const int rnd = unit_size * 8 - 1, shift = unit_size_log2 + 3;
 2692|  21.5k|                const int x0 = ((4 *  t->bx            * d >> ss_hor) + rnd) >> shift;
 2693|  21.5k|                const int x1 = ((4 * (t->bx + sb_step) * d >> ss_hor) + rnd) >> shift;
 2694|       |
 2695|  46.3k|                for (int x = x0; x < imin(x1, n_units); x++) {
  ------------------
  |  Branch (2695:34): [True: 24.8k, False: 21.5k]
  ------------------
 2696|  24.8k|                    const int px_x = x << (unit_size_log2 + ss_hor);
 2697|  24.8k|                    const int sb_idx = (t->by >> 5) * f->sr_sb128w + (px_x >> 7);
 2698|  24.8k|                    const int unit_idx = ((t->by & 16) >> 3) + ((px_x & 64) >> 6);
 2699|  24.8k|                    Av1RestorationUnit *const lr = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2700|       |
 2701|  24.8k|                    read_restoration_info(t, lr, p, frame_type);
 2702|  24.8k|                }
 2703|  95.8k|            } else {
 2704|  95.8k|                const int x = 4 * t->bx >> ss_hor;
 2705|  95.8k|                if (x & mask) continue;
  ------------------
  |  Branch (2705:21): [True: 10.6k, False: 85.1k]
  ------------------
 2706|  85.1k|                const int w = (f->cur.p.w + ss_hor) >> ss_hor;
 2707|       |                // Round half up at frame boundaries, if there's more than one
 2708|       |                // restoration unit
 2709|  85.1k|                if (x && x + half_unit > w) continue;
  ------------------
  |  Branch (2709:21): [True: 56.3k, False: 28.8k]
  |  Branch (2709:26): [True: 1.03k, False: 55.3k]
  ------------------
 2710|  84.1k|                const int sb_idx = (t->by >> 5) * f->sr_sb128w + (t->bx >> 5);
 2711|  84.1k|                const int unit_idx = ((t->by & 16) >> 3) + ((t->bx & 16) >> 4);
 2712|  84.1k|                Av1RestorationUnit *const lr = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2713|       |
 2714|  84.1k|                read_restoration_info(t, lr, p, frame_type);
 2715|  84.1k|            }
 2716|   117k|        }
 2717|   601k|        if (decode_sb(t, root_bl, dav1d_intra_edge_tree[root_bl]))
  ------------------
  |  Branch (2717:13): [True: 9.08k, False: 592k]
  ------------------
 2718|  9.08k|            return 1;
 2719|   592k|        if (t->bx & 16 || f->seq_hdr->sb128) {
  ------------------
  |  Branch (2719:13): [True: 174k, False: 418k]
  |  Branch (2719:27): [True: 227k, False: 190k]
  ------------------
 2720|   402k|            t->a++;
 2721|   402k|            t->lf_mask++;
 2722|   402k|        }
 2723|   592k|    }
 2724|       |
 2725|   170k|    if (f->seq_hdr->ref_frame_mvs && f->c->n_tc > 1 && IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (2725:9): [True: 112k, False: 57.3k]
  |  Branch (2725:38): [True: 0, False: 112k]
  ------------------
 2726|      0|        dav1d_refmvs_save_tmvs(&f->c->refmvs_dsp, &t->rt,
 2727|      0|                               ts->tiling.col_start >> 1, ts->tiling.col_end >> 1,
 2728|      0|                               t->by >> 1, (t->by + sb_step) >> 1);
 2729|      0|    }
 2730|       |
 2731|       |    // backup pre-loopfilter pixels for intra prediction of the next sbrow
 2732|   170k|    if (t->frame_thread.pass != 1)
  ------------------
  |  Branch (2732:9): [True: 170k, False: 0]
  ------------------
 2733|   170k|        f->bd_fn.backup_ipred_edge(t);
 2734|       |
 2735|       |    // backup t->a/l.tx_lpf_y/uv at tile boundaries to use them to "fix"
 2736|       |    // up the initial value in neighbour tiles when running the loopfilter
 2737|   170k|    int align_h = (f->bh + 31) & ~31;
 2738|   170k|    memcpy(&f->lf.tx_lpf_right_edge[0][align_h * tile_col + t->by],
 2739|   170k|           &t->l.tx_lpf_y[t->by & 16], sb_step);
 2740|   170k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2741|   170k|    align_h >>= ss_ver;
 2742|   170k|    memcpy(&f->lf.tx_lpf_right_edge[1][align_h * tile_col + (t->by >> ss_ver)],
 2743|   170k|           &t->l.tx_lpf_uv[(t->by & 16) >> ss_ver], sb_step >> ss_ver);
 2744|       |
 2745|       |    // error out on symbol decoder overread
 2746|   170k|    if (ts->msac.cnt <= -15) return 1;
  ------------------
  |  Branch (2746:9): [True: 15.0k, False: 155k]
  ------------------
 2747|       |
 2748|   155k|    return c->strict_std_compliance &&
  ------------------
  |  Branch (2748:12): [True: 0, False: 155k]
  ------------------
 2749|      0|           (t->by >> f->sb_shift) + 1 >= f->frame_hdr->tiling.row_start_sb[tile_row + 1] &&
  ------------------
  |  Branch (2749:12): [True: 0, False: 0]
  ------------------
 2750|      0|           check_trailing_bits_after_symbol_coder(&ts->msac);
  ------------------
  |  Branch (2750:12): [True: 0, False: 0]
  ------------------
 2751|   170k|}
dav1d_decode_frame_init:
 2753|  45.8k|int dav1d_decode_frame_init(Dav1dFrameContext *const f) {
 2754|  45.8k|    const Dav1dContext *const c = f->c;
 2755|  45.8k|    int retval = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|  45.8k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 2756|       |
 2757|  45.8k|    if (f->sbh > f->lf.start_of_tile_row_sz) {
  ------------------
  |  Branch (2757:9): [True: 9.04k, False: 36.7k]
  ------------------
 2758|  9.04k|        dav1d_free(f->lf.start_of_tile_row);
  ------------------
  |  |  135|  9.04k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2759|  9.04k|        f->lf.start_of_tile_row = dav1d_malloc(ALLOC_TILE, f->sbh * sizeof(uint8_t));
  ------------------
  |  |  132|  9.04k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2760|  9.04k|        if (!f->lf.start_of_tile_row) {
  ------------------
  |  Branch (2760:13): [True: 0, False: 9.04k]
  ------------------
 2761|      0|            f->lf.start_of_tile_row_sz = 0;
 2762|      0|            goto error;
 2763|      0|        }
 2764|  9.04k|        f->lf.start_of_tile_row_sz = f->sbh;
 2765|  9.04k|    }
 2766|  45.8k|    int sby = 0;
 2767|  95.5k|    for (int tile_row = 0; tile_row < f->frame_hdr->tiling.rows; tile_row++) {
  ------------------
  |  Branch (2767:28): [True: 49.7k, False: 45.8k]
  ------------------
 2768|  49.7k|        f->lf.start_of_tile_row[sby++] = tile_row;
 2769|   443k|        while (sby < f->frame_hdr->tiling.row_start_sb[tile_row + 1])
  ------------------
  |  Branch (2769:16): [True: 394k, False: 49.7k]
  ------------------
 2770|   394k|            f->lf.start_of_tile_row[sby++] = 0;
 2771|  49.7k|    }
 2772|       |
 2773|  45.8k|    const int n_ts = f->frame_hdr->tiling.cols * f->frame_hdr->tiling.rows;
 2774|  45.8k|    if (n_ts != f->n_ts) {
  ------------------
  |  Branch (2774:9): [True: 9.56k, False: 36.2k]
  ------------------
 2775|  9.56k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (2775:13): [True: 0, False: 9.56k]
  ------------------
 2776|      0|            dav1d_free(f->frame_thread.tile_start_off);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
 2777|      0|            f->frame_thread.tile_start_off =
 2778|      0|                dav1d_malloc(ALLOC_TILE, sizeof(*f->frame_thread.tile_start_off) * n_ts);
  ------------------
  |  |  132|      0|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2779|      0|            if (!f->frame_thread.tile_start_off) {
  ------------------
  |  Branch (2779:17): [True: 0, False: 0]
  ------------------
 2780|      0|                f->n_ts = 0;
 2781|      0|                goto error;
 2782|      0|            }
 2783|      0|        }
 2784|  9.56k|        dav1d_free_aligned(f->ts);
  ------------------
  |  |  136|  9.56k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2785|  9.56k|        f->ts = dav1d_alloc_aligned(ALLOC_TILE, sizeof(*f->ts) * n_ts, 32);
  ------------------
  |  |  134|  9.56k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2786|  9.56k|        if (!f->ts) goto error;
  ------------------
  |  Branch (2786:13): [True: 0, False: 9.56k]
  ------------------
 2787|  9.56k|        f->n_ts = n_ts;
 2788|  9.56k|    }
 2789|       |
 2790|  45.8k|    const int a_sz = f->sb128w * f->frame_hdr->tiling.rows * (1 + (c->n_fc > 1 && c->n_tc > 1));
  ------------------
  |  Branch (2790:68): [True: 0, False: 45.8k]
  |  Branch (2790:83): [True: 0, False: 0]
  ------------------
 2791|  45.8k|    if (a_sz != f->a_sz) {
  ------------------
  |  Branch (2791:9): [True: 10.5k, False: 35.3k]
  ------------------
 2792|  10.5k|        dav1d_free(f->a);
  ------------------
  |  |  135|  10.5k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2793|  10.5k|        f->a = dav1d_malloc(ALLOC_TILE, sizeof(*f->a) * a_sz);
  ------------------
  |  |  132|  10.5k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2794|  10.5k|        if (!f->a) {
  ------------------
  |  Branch (2794:13): [True: 0, False: 10.5k]
  ------------------
 2795|      0|            f->a_sz = 0;
 2796|      0|            goto error;
 2797|      0|        }
 2798|  10.5k|        f->a_sz = a_sz;
 2799|  10.5k|    }
 2800|       |
 2801|  45.8k|    const int num_sb128 = f->sb128w * f->sb128h;
 2802|  45.8k|    const uint8_t *const size_mul = ss_size_mul[f->cur.p.layout];
 2803|  45.8k|    const int hbd = !!f->seq_hdr->hbd;
 2804|  45.8k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (2804:9): [True: 0, False: 45.8k]
  ------------------
 2805|      0|        const unsigned sb_step4 = f->sb_step * 4;
 2806|      0|        int tile_idx = 0;
 2807|      0|        for (int tile_row = 0; tile_row < f->frame_hdr->tiling.rows; tile_row++) {
  ------------------
  |  Branch (2807:32): [True: 0, False: 0]
  ------------------
 2808|      0|            const unsigned row_off = f->frame_hdr->tiling.row_start_sb[tile_row] *
 2809|      0|                                     sb_step4 * f->sb128w * 128;
 2810|      0|            const unsigned b_diff = (f->frame_hdr->tiling.row_start_sb[tile_row + 1] -
 2811|      0|                                     f->frame_hdr->tiling.row_start_sb[tile_row]) * sb_step4;
 2812|      0|            for (int tile_col = 0; tile_col < f->frame_hdr->tiling.cols; tile_col++) {
  ------------------
  |  Branch (2812:36): [True: 0, False: 0]
  ------------------
 2813|      0|                f->frame_thread.tile_start_off[tile_idx++] = row_off + b_diff *
 2814|      0|                    f->frame_hdr->tiling.col_start_sb[tile_col] * sb_step4;
 2815|      0|            }
 2816|      0|        }
 2817|       |
 2818|      0|        const int lowest_pixel_mem_sz = f->frame_hdr->tiling.cols * f->sbh;
 2819|      0|        if (lowest_pixel_mem_sz != f->tile_thread.lowest_pixel_mem_sz) {
  ------------------
  |  Branch (2819:13): [True: 0, False: 0]
  ------------------
 2820|      0|            dav1d_free(f->tile_thread.lowest_pixel_mem);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
 2821|      0|            f->tile_thread.lowest_pixel_mem =
 2822|      0|                dav1d_malloc(ALLOC_TILE, lowest_pixel_mem_sz *
  ------------------
  |  |  132|      0|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2823|      0|                             sizeof(*f->tile_thread.lowest_pixel_mem));
 2824|      0|            if (!f->tile_thread.lowest_pixel_mem) {
  ------------------
  |  Branch (2824:17): [True: 0, False: 0]
  ------------------
 2825|      0|                f->tile_thread.lowest_pixel_mem_sz = 0;
 2826|      0|                goto error;
 2827|      0|            }
 2828|      0|            f->tile_thread.lowest_pixel_mem_sz = lowest_pixel_mem_sz;
 2829|      0|        }
 2830|      0|        int (*lowest_pixel_ptr)[7][2] = f->tile_thread.lowest_pixel_mem;
 2831|      0|        for (int tile_row = 0, tile_row_base = 0; tile_row < f->frame_hdr->tiling.rows;
  ------------------
  |  Branch (2831:51): [True: 0, False: 0]
  ------------------
 2832|      0|             tile_row++, tile_row_base += f->frame_hdr->tiling.cols)
 2833|      0|        {
 2834|      0|            const int tile_row_sb_h = f->frame_hdr->tiling.row_start_sb[tile_row + 1] -
 2835|      0|                                      f->frame_hdr->tiling.row_start_sb[tile_row];
 2836|      0|            for (int tile_col = 0; tile_col < f->frame_hdr->tiling.cols; tile_col++) {
  ------------------
  |  Branch (2836:36): [True: 0, False: 0]
  ------------------
 2837|      0|                f->ts[tile_row_base + tile_col].lowest_pixel = lowest_pixel_ptr;
 2838|      0|                lowest_pixel_ptr += tile_row_sb_h;
 2839|      0|            }
 2840|      0|        }
 2841|       |
 2842|      0|        const int cbi_sz = num_sb128 * size_mul[0];
 2843|      0|        if (cbi_sz != f->frame_thread.cbi_sz) {
  ------------------
  |  Branch (2843:13): [True: 0, False: 0]
  ------------------
 2844|      0|            dav1d_free_aligned(f->frame_thread.cbi);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2845|      0|            f->frame_thread.cbi =
 2846|      0|                dav1d_alloc_aligned(ALLOC_BLOCK, sizeof(*f->frame_thread.cbi) *
  ------------------
  |  |  134|      0|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2847|      0|                                    cbi_sz * 32 * 32 / 4, 64);
 2848|      0|            if (!f->frame_thread.cbi) {
  ------------------
  |  Branch (2848:17): [True: 0, False: 0]
  ------------------
 2849|      0|                f->frame_thread.cbi_sz = 0;
 2850|      0|                goto error;
 2851|      0|            }
 2852|      0|            f->frame_thread.cbi_sz = cbi_sz;
 2853|      0|        }
 2854|       |
 2855|      0|        const int cf_sz = (num_sb128 * size_mul[0]) << hbd;
 2856|      0|        if (cf_sz != f->frame_thread.cf_sz) {
  ------------------
  |  Branch (2856:13): [True: 0, False: 0]
  ------------------
 2857|      0|            dav1d_free_aligned(f->frame_thread.cf);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2858|      0|            f->frame_thread.cf =
 2859|      0|                dav1d_alloc_aligned(ALLOC_COEF, (size_t)cf_sz * 128 * 128 / 2, 64);
  ------------------
  |  |  134|      0|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2860|      0|            if (!f->frame_thread.cf) {
  ------------------
  |  Branch (2860:17): [True: 0, False: 0]
  ------------------
 2861|      0|                f->frame_thread.cf_sz = 0;
 2862|      0|                goto error;
 2863|      0|            }
 2864|      0|            memset(f->frame_thread.cf, 0, (size_t)cf_sz * 128 * 128 / 2);
 2865|      0|            f->frame_thread.cf_sz = cf_sz;
 2866|      0|        }
 2867|       |
 2868|      0|        if (f->frame_hdr->allow_screen_content_tools) {
  ------------------
  |  Branch (2868:13): [True: 0, False: 0]
  ------------------
 2869|      0|            const int pal_sz = num_sb128 << hbd;
 2870|      0|            if (pal_sz != f->frame_thread.pal_sz) {
  ------------------
  |  Branch (2870:17): [True: 0, False: 0]
  ------------------
 2871|      0|                dav1d_free_aligned(f->frame_thread.pal);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2872|      0|                f->frame_thread.pal =
 2873|      0|                    dav1d_alloc_aligned(ALLOC_PAL, sizeof(*f->frame_thread.pal) *
  ------------------
  |  |  134|      0|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2874|      0|                                        pal_sz * 16 * 16, 64);
 2875|      0|                if (!f->frame_thread.pal) {
  ------------------
  |  Branch (2875:21): [True: 0, False: 0]
  ------------------
 2876|      0|                    f->frame_thread.pal_sz = 0;
 2877|      0|                    goto error;
 2878|      0|                }
 2879|      0|                f->frame_thread.pal_sz = pal_sz;
 2880|      0|            }
 2881|       |
 2882|      0|            const int pal_idx_sz = num_sb128 * size_mul[1];
 2883|      0|            if (pal_idx_sz != f->frame_thread.pal_idx_sz) {
  ------------------
  |  Branch (2883:17): [True: 0, False: 0]
  ------------------
 2884|      0|                dav1d_free_aligned(f->frame_thread.pal_idx);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2885|      0|                f->frame_thread.pal_idx =
 2886|      0|                    dav1d_alloc_aligned(ALLOC_PAL, sizeof(*f->frame_thread.pal_idx) *
  ------------------
  |  |  134|      0|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2887|      0|                                        pal_idx_sz * 128 * 128 / 8, 64);
 2888|      0|                if (!f->frame_thread.pal_idx) {
  ------------------
  |  Branch (2888:21): [True: 0, False: 0]
  ------------------
 2889|      0|                    f->frame_thread.pal_idx_sz = 0;
 2890|      0|                    goto error;
 2891|      0|                }
 2892|      0|                f->frame_thread.pal_idx_sz = pal_idx_sz;
 2893|      0|            }
 2894|      0|        } else if (f->frame_thread.pal) {
  ------------------
  |  Branch (2894:20): [True: 0, False: 0]
  ------------------
 2895|      0|            dav1d_freep_aligned(&f->frame_thread.pal);
 2896|      0|            dav1d_freep_aligned(&f->frame_thread.pal_idx);
 2897|      0|            f->frame_thread.pal_sz = f->frame_thread.pal_idx_sz = 0;
 2898|      0|        }
 2899|      0|    }
 2900|       |
 2901|       |    // update allocation of block contexts for above
 2902|  45.8k|    ptrdiff_t y_stride = f->cur.stride[0], uv_stride = f->cur.stride[1];
 2903|  45.8k|    const int has_resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
 2904|  45.8k|    const int need_cdef_lpf_copy = c->n_tc > 1 && has_resize;
  ------------------
  |  Branch (2904:36): [True: 0, False: 45.8k]
  |  Branch (2904:51): [True: 0, False: 0]
  ------------------
 2905|  45.8k|    if (y_stride * f->sbh * 4 != f->lf.cdef_buf_plane_sz[0] ||
  ------------------
  |  Branch (2905:9): [True: 10.1k, False: 35.6k]
  ------------------
 2906|  35.6k|        uv_stride * f->sbh * 8 != f->lf.cdef_buf_plane_sz[1] ||
  ------------------
  |  Branch (2906:9): [True: 1.09k, False: 34.5k]
  ------------------
 2907|  34.5k|        need_cdef_lpf_copy != f->lf.need_cdef_lpf_copy ||
  ------------------
  |  Branch (2907:9): [True: 0, False: 34.5k]
  ------------------
 2908|  34.5k|        f->sbh != f->lf.cdef_buf_sbh)
  ------------------
  |  Branch (2908:9): [True: 464, False: 34.1k]
  ------------------
 2909|  11.7k|    {
 2910|  11.7k|        dav1d_free_aligned(f->lf.cdef_line_buf);
  ------------------
  |  |  136|  11.7k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2911|  11.7k|        size_t alloc_sz = 64;
 2912|  11.7k|        alloc_sz += (size_t)llabs(y_stride) * 4 * f->sbh << need_cdef_lpf_copy;
 2913|  11.7k|        alloc_sz += (size_t)llabs(uv_stride) * 8 * f->sbh << need_cdef_lpf_copy;
 2914|  11.7k|        uint8_t *ptr = f->lf.cdef_line_buf = dav1d_alloc_aligned(ALLOC_CDEF, alloc_sz, 32);
  ------------------
  |  |  134|  11.7k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2915|  11.7k|        if (!ptr) {
  ------------------
  |  Branch (2915:13): [True: 0, False: 11.7k]
  ------------------
 2916|      0|            f->lf.cdef_buf_plane_sz[0] = f->lf.cdef_buf_plane_sz[1] = 0;
 2917|      0|            goto error;
 2918|      0|        }
 2919|       |
 2920|  11.7k|        ptr += 32;
 2921|  11.7k|        if (y_stride < 0) {
  ------------------
  |  Branch (2921:13): [True: 0, False: 11.7k]
  ------------------
 2922|      0|            f->lf.cdef_line[0][0] = ptr - y_stride * (f->sbh * 4 - 1);
 2923|      0|            f->lf.cdef_line[1][0] = ptr - y_stride * (f->sbh * 4 - 3);
 2924|  11.7k|        } else {
 2925|  11.7k|            f->lf.cdef_line[0][0] = ptr + y_stride * 0;
 2926|  11.7k|            f->lf.cdef_line[1][0] = ptr + y_stride * 2;
 2927|  11.7k|        }
 2928|  11.7k|        ptr += llabs(y_stride) * f->sbh * 4;
 2929|  11.7k|        if (uv_stride < 0) {
  ------------------
  |  Branch (2929:13): [True: 0, False: 11.7k]
  ------------------
 2930|      0|            f->lf.cdef_line[0][1] = ptr - uv_stride * (f->sbh * 8 - 1);
 2931|      0|            f->lf.cdef_line[0][2] = ptr - uv_stride * (f->sbh * 8 - 3);
 2932|      0|            f->lf.cdef_line[1][1] = ptr - uv_stride * (f->sbh * 8 - 5);
 2933|      0|            f->lf.cdef_line[1][2] = ptr - uv_stride * (f->sbh * 8 - 7);
 2934|  11.7k|        } else {
 2935|  11.7k|            f->lf.cdef_line[0][1] = ptr + uv_stride * 0;
 2936|  11.7k|            f->lf.cdef_line[0][2] = ptr + uv_stride * 2;
 2937|  11.7k|            f->lf.cdef_line[1][1] = ptr + uv_stride * 4;
 2938|  11.7k|            f->lf.cdef_line[1][2] = ptr + uv_stride * 6;
 2939|  11.7k|        }
 2940|       |
 2941|  11.7k|        if (need_cdef_lpf_copy) {
  ------------------
  |  Branch (2941:13): [True: 0, False: 11.7k]
  ------------------
 2942|      0|            ptr += llabs(uv_stride) * f->sbh * 8;
 2943|      0|            if (y_stride < 0)
  ------------------
  |  Branch (2943:17): [True: 0, False: 0]
  ------------------
 2944|      0|                f->lf.cdef_lpf_line[0] = ptr - y_stride * (f->sbh * 4 - 1);
 2945|      0|            else
 2946|      0|                f->lf.cdef_lpf_line[0] = ptr;
 2947|      0|            ptr += llabs(y_stride) * f->sbh * 4;
 2948|      0|            if (uv_stride < 0) {
  ------------------
  |  Branch (2948:17): [True: 0, False: 0]
  ------------------
 2949|      0|                f->lf.cdef_lpf_line[1] = ptr - uv_stride * (f->sbh * 4 - 1);
 2950|      0|                f->lf.cdef_lpf_line[2] = ptr - uv_stride * (f->sbh * 8 - 1);
 2951|      0|            } else {
 2952|      0|                f->lf.cdef_lpf_line[1] = ptr;
 2953|      0|                f->lf.cdef_lpf_line[2] = ptr + uv_stride * f->sbh * 4;
 2954|      0|            }
 2955|      0|        }
 2956|       |
 2957|  11.7k|        f->lf.cdef_buf_plane_sz[0] = (int) y_stride * f->sbh * 4;
 2958|  11.7k|        f->lf.cdef_buf_plane_sz[1] = (int) uv_stride * f->sbh * 8;
 2959|  11.7k|        f->lf.need_cdef_lpf_copy = need_cdef_lpf_copy;
 2960|  11.7k|        f->lf.cdef_buf_sbh = f->sbh;
 2961|  11.7k|    }
 2962|       |
 2963|  45.8k|    const int sb128 = f->seq_hdr->sb128;
 2964|  45.8k|    const int num_lines = c->n_tc > 1 ? f->sbh * 4 << sb128 : 12;
  ------------------
  |  Branch (2964:27): [True: 0, False: 45.8k]
  ------------------
 2965|  45.8k|    y_stride = f->sr_cur.p.stride[0], uv_stride = f->sr_cur.p.stride[1];
 2966|  45.8k|    if (y_stride * num_lines != f->lf.lr_buf_plane_sz[0] ||
  ------------------
  |  Branch (2966:9): [True: 10.2k, False: 35.5k]
  ------------------
 2967|  35.5k|        uv_stride * num_lines * 2 != f->lf.lr_buf_plane_sz[1])
  ------------------
  |  Branch (2967:9): [True: 1.09k, False: 34.4k]
  ------------------
 2968|  11.3k|    {
 2969|  11.3k|        dav1d_free_aligned(f->lf.lr_line_buf);
  ------------------
  |  |  136|  11.3k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2970|       |        // lr simd may overread the input, so slightly over-allocate the lpf buffer
 2971|  11.3k|        size_t alloc_sz = 128;
 2972|  11.3k|        alloc_sz += (size_t)llabs(y_stride) * num_lines;
 2973|  11.3k|        alloc_sz += (size_t)llabs(uv_stride) * num_lines * 2;
 2974|  11.3k|        uint8_t *ptr = f->lf.lr_line_buf = dav1d_alloc_aligned(ALLOC_LR, alloc_sz, 64);
  ------------------
  |  |  134|  11.3k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2975|  11.3k|        if (!ptr) {
  ------------------
  |  Branch (2975:13): [True: 0, False: 11.3k]
  ------------------
 2976|      0|            f->lf.lr_buf_plane_sz[0] = f->lf.lr_buf_plane_sz[1] = 0;
 2977|      0|            goto error;
 2978|      0|        }
 2979|       |
 2980|  11.3k|        ptr += 64;
 2981|  11.3k|        if (y_stride < 0)
  ------------------
  |  Branch (2981:13): [True: 0, False: 11.3k]
  ------------------
 2982|      0|            f->lf.lr_lpf_line[0] = ptr - y_stride * (num_lines - 1);
 2983|  11.3k|        else
 2984|  11.3k|            f->lf.lr_lpf_line[0] = ptr;
 2985|  11.3k|        ptr += llabs(y_stride) * num_lines;
 2986|  11.3k|        if (uv_stride < 0) {
  ------------------
  |  Branch (2986:13): [True: 0, False: 11.3k]
  ------------------
 2987|      0|            f->lf.lr_lpf_line[1] = ptr - uv_stride * (num_lines * 1 - 1);
 2988|      0|            f->lf.lr_lpf_line[2] = ptr - uv_stride * (num_lines * 2 - 1);
 2989|  11.3k|        } else {
 2990|  11.3k|            f->lf.lr_lpf_line[1] = ptr;
 2991|  11.3k|            f->lf.lr_lpf_line[2] = ptr + uv_stride * num_lines;
 2992|  11.3k|        }
 2993|       |
 2994|  11.3k|        f->lf.lr_buf_plane_sz[0] = (int) y_stride * num_lines;
 2995|  11.3k|        f->lf.lr_buf_plane_sz[1] = (int) uv_stride * num_lines * 2;
 2996|  11.3k|    }
 2997|       |
 2998|       |    // update allocation for loopfilter masks
 2999|  45.8k|    if (num_sb128 != f->lf.mask_sz) {
  ------------------
  |  Branch (2999:9): [True: 10.1k, False: 35.7k]
  ------------------
 3000|  10.1k|        dav1d_free(f->lf.mask);
  ------------------
  |  |  135|  10.1k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3001|  10.1k|        dav1d_free(f->lf.level);
  ------------------
  |  |  135|  10.1k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3002|  10.1k|        f->lf.mask = dav1d_malloc(ALLOC_LF, sizeof(*f->lf.mask) * num_sb128);
  ------------------
  |  |  132|  10.1k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3003|       |        // over-allocate by 3 bytes since some of the SIMD implementations
 3004|       |        // index this from the level type and can thus over-read by up to 3
 3005|  10.1k|        f->lf.level = dav1d_malloc(ALLOC_LF, sizeof(*f->lf.level) * num_sb128 * 32 * 32 + 3);
  ------------------
  |  |  132|  10.1k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3006|  10.1k|        if (!f->lf.mask || !f->lf.level) {
  ------------------
  |  Branch (3006:13): [True: 0, False: 10.1k]
  |  Branch (3006:28): [True: 0, False: 10.1k]
  ------------------
 3007|      0|            f->lf.mask_sz = 0;
 3008|      0|            goto error;
 3009|      0|        }
 3010|  10.1k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (3010:13): [True: 0, False: 10.1k]
  ------------------
 3011|      0|            dav1d_free(f->frame_thread.b);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
 3012|      0|            f->frame_thread.b = dav1d_malloc(ALLOC_BLOCK, sizeof(*f->frame_thread.b) *
  ------------------
  |  |  132|      0|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3013|      0|                                             num_sb128 * 32 * 32);
 3014|      0|            if (!f->frame_thread.b) {
  ------------------
  |  Branch (3014:17): [True: 0, False: 0]
  ------------------
 3015|      0|                f->lf.mask_sz = 0;
 3016|      0|                goto error;
 3017|      0|            }
 3018|      0|        }
 3019|  10.1k|        f->lf.mask_sz = num_sb128;
 3020|  10.1k|    }
 3021|       |
 3022|  45.8k|    f->sr_sb128w = (f->sr_cur.p.p.w + 127) >> 7;
 3023|  45.8k|    const int lr_mask_sz = f->sr_sb128w * f->sb128h;
 3024|  45.8k|    if (lr_mask_sz != f->lf.lr_mask_sz) {
  ------------------
  |  Branch (3024:9): [True: 9.96k, False: 35.8k]
  ------------------
 3025|  9.96k|        dav1d_free(f->lf.lr_mask);
  ------------------
  |  |  135|  9.96k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3026|  9.96k|        f->lf.lr_mask = dav1d_malloc(ALLOC_LR, sizeof(*f->lf.lr_mask) * lr_mask_sz);
  ------------------
  |  |  132|  9.96k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3027|  9.96k|        if (!f->lf.lr_mask) {
  ------------------
  |  Branch (3027:13): [True: 0, False: 9.96k]
  ------------------
 3028|      0|            f->lf.lr_mask_sz = 0;
 3029|      0|            goto error;
 3030|      0|        }
 3031|  9.96k|        f->lf.lr_mask_sz = lr_mask_sz;
 3032|  9.96k|    }
 3033|  45.8k|    f->lf.restore_planes =
 3034|  45.8k|        ((f->frame_hdr->restoration.type[0] != DAV1D_RESTORATION_NONE) << 0) +
 3035|  45.8k|        ((f->frame_hdr->restoration.type[1] != DAV1D_RESTORATION_NONE) << 1) +
 3036|  45.8k|        ((f->frame_hdr->restoration.type[2] != DAV1D_RESTORATION_NONE) << 2);
 3037|  45.8k|    if (f->frame_hdr->loopfilter.sharpness != f->lf.last_sharpness) {
  ------------------
  |  Branch (3037:9): [True: 16.0k, False: 29.7k]
  ------------------
 3038|  16.0k|        dav1d_calc_eih(&f->lf.lim_lut, f->frame_hdr->loopfilter.sharpness);
 3039|  16.0k|        f->lf.last_sharpness = f->frame_hdr->loopfilter.sharpness;
 3040|  16.0k|    }
 3041|  45.8k|    dav1d_calc_lf_values(f->lf.lvl, f->frame_hdr, (int8_t[4]) { 0, 0, 0, 0 });
 3042|  45.8k|    memset(f->lf.mask, 0, sizeof(*f->lf.mask) * num_sb128);
 3043|       |
 3044|  45.8k|    const int ipred_edge_sz = f->sbh * f->sb128w << hbd;
 3045|  45.8k|    if (ipred_edge_sz != f->ipred_edge_sz) {
  ------------------
  |  Branch (3045:9): [True: 10.1k, False: 35.6k]
  ------------------
 3046|  10.1k|        dav1d_free_aligned(f->ipred_edge[0]);
  ------------------
  |  |  136|  10.1k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 3047|  10.1k|        uint8_t *ptr = f->ipred_edge[0] =
 3048|  10.1k|            dav1d_alloc_aligned(ALLOC_IPRED, ipred_edge_sz * 128 * 3, 64);
  ------------------
  |  |  134|  10.1k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 3049|  10.1k|        if (!ptr) {
  ------------------
  |  Branch (3049:13): [True: 0, False: 10.1k]
  ------------------
 3050|      0|            f->ipred_edge_sz = 0;
 3051|      0|            goto error;
 3052|      0|        }
 3053|  10.1k|        f->ipred_edge[1] = ptr + ipred_edge_sz * 128 * 1;
 3054|  10.1k|        f->ipred_edge[2] = ptr + ipred_edge_sz * 128 * 2;
 3055|  10.1k|        f->ipred_edge_sz = ipred_edge_sz;
 3056|  10.1k|    }
 3057|       |
 3058|  45.8k|    const int re_sz = f->sb128h * f->frame_hdr->tiling.cols;
 3059|  45.8k|    if (re_sz != f->lf.re_sz) {
  ------------------
  |  Branch (3059:9): [True: 9.55k, False: 36.2k]
  ------------------
 3060|  9.55k|        dav1d_free(f->lf.tx_lpf_right_edge[0]);
  ------------------
  |  |  135|  9.55k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3061|  9.55k|        f->lf.tx_lpf_right_edge[0] = dav1d_malloc(ALLOC_LF, re_sz * 32 * 2);
  ------------------
  |  |  132|  9.55k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3062|  9.55k|        if (!f->lf.tx_lpf_right_edge[0]) {
  ------------------
  |  Branch (3062:13): [True: 0, False: 9.55k]
  ------------------
 3063|      0|            f->lf.re_sz = 0;
 3064|      0|            goto error;
 3065|      0|        }
 3066|  9.55k|        f->lf.tx_lpf_right_edge[1] = f->lf.tx_lpf_right_edge[0] + re_sz * 32;
 3067|  9.55k|        f->lf.re_sz = re_sz;
 3068|  9.55k|    }
 3069|       |
 3070|       |    // init ref mvs
 3071|  45.8k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|  91.6k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 13.4k, False: 32.3k]
  |  |  ------------------
  ------------------
  |  Branch (3071:45): [True: 17.8k, False: 14.4k]
  ------------------
 3072|  31.3k|        const int ret =
 3073|  31.3k|            dav1d_refmvs_init_frame(&f->rf, f->seq_hdr, f->frame_hdr,
 3074|  31.3k|                                    f->refpoc, f->mvs, f->refrefpoc, f->ref_mvs,
 3075|  31.3k|                                    f->c->n_tc, f->c->n_fc);
 3076|  31.3k|        if (ret < 0) goto error;
  ------------------
  |  Branch (3076:13): [True: 0, False: 31.3k]
  ------------------
 3077|  31.3k|    }
 3078|       |
 3079|       |    // setup dequant tables
 3080|  45.8k|    init_quant_tables(f->seq_hdr, f->frame_hdr, f->frame_hdr->quant.yac, f->dq);
 3081|  45.8k|    if (f->frame_hdr->quant.qm)
  ------------------
  |  Branch (3081:9): [True: 6.50k, False: 39.3k]
  ------------------
 3082|   130k|        for (int i = 0; i < N_RECT_TX_SIZES; i++) {
  ------------------
  |  Branch (3082:25): [True: 123k, False: 6.50k]
  ------------------
 3083|   123k|            f->qm[i][0] = dav1d_qm_tbl[f->frame_hdr->quant.qm_y][0][i];
 3084|   123k|            f->qm[i][1] = dav1d_qm_tbl[f->frame_hdr->quant.qm_u][1][i];
 3085|   123k|            f->qm[i][2] = dav1d_qm_tbl[f->frame_hdr->quant.qm_v][1][i];
 3086|   123k|        }
 3087|  39.3k|    else
 3088|  39.3k|        memset(f->qm, 0, sizeof(f->qm));
 3089|       |
 3090|       |    // setup jnt_comp weights
 3091|  45.8k|    if (f->frame_hdr->switchable_comp_refs) {
  ------------------
  |  Branch (3091:9): [True: 10.2k, False: 35.5k]
  ------------------
 3092|  81.9k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3092:25): [True: 71.7k, False: 10.2k]
  ------------------
 3093|  71.7k|            const unsigned ref0poc = f->refp[i].p.frame_hdr->frame_offset;
 3094|       |
 3095|   286k|            for (int j = i + 1; j < 7; j++) {
  ------------------
  |  Branch (3095:33): [True: 215k, False: 71.7k]
  ------------------
 3096|   215k|                const unsigned ref1poc = f->refp[j].p.frame_hdr->frame_offset;
 3097|       |
 3098|   215k|                const unsigned d1 =
 3099|   215k|                    imin(abs(get_poc_diff(f->seq_hdr->order_hint_n_bits, ref0poc,
 3100|   215k|                                          f->cur.frame_hdr->frame_offset)), 31);
 3101|   215k|                const unsigned d0 =
 3102|   215k|                    imin(abs(get_poc_diff(f->seq_hdr->order_hint_n_bits, ref1poc,
 3103|   215k|                                          f->cur.frame_hdr->frame_offset)), 31);
 3104|   215k|                const int order = d0 <= d1;
 3105|       |
 3106|   215k|                static const uint8_t quant_dist_weight[3][2] = {
 3107|   215k|                    { 2, 3 }, { 2, 5 }, { 2, 7 }
 3108|   215k|                };
 3109|   215k|                static const uint8_t quant_dist_lookup_table[4][2] = {
 3110|   215k|                    { 9, 7 }, { 11, 5 }, { 12, 4 }, { 13, 3 }
 3111|   215k|                };
 3112|       |
 3113|   215k|                int k;
 3114|   718k|                for (k = 0; k < 3; k++) {
  ------------------
  |  Branch (3114:29): [True: 553k, False: 164k]
  ------------------
 3115|   553k|                    const int c0 = quant_dist_weight[k][order];
 3116|   553k|                    const int c1 = quant_dist_weight[k][!order];
 3117|   553k|                    const int d0_c0 = d0 * c0;
 3118|   553k|                    const int d1_c1 = d1 * c1;
 3119|   553k|                    if ((d0 > d1 && d0_c0 < d1_c1) || (d0 <= d1 && d0_c0 > d1_c1)) break;
  ------------------
  |  Branch (3119:26): [True: 102k, False: 450k]
  |  Branch (3119:37): [True: 5.69k, False: 96.6k]
  |  Branch (3119:56): [True: 450k, False: 96.6k]
  |  Branch (3119:68): [True: 44.5k, False: 406k]
  ------------------
 3120|   553k|                }
 3121|       |
 3122|   215k|                f->jnt_weights[i][j] = quant_dist_lookup_table[k][order];
 3123|   215k|            }
 3124|  71.7k|        }
 3125|  10.2k|    }
 3126|       |
 3127|       |    /* Init loopfilter pointers. Increasing NULL pointers is technically UB,
 3128|       |     * so just point the chroma pointers in 4:0:0 to the luma plane here to
 3129|       |     * avoid having additional in-loop branches in various places. We never
 3130|       |     * dereference those pointers so it doesn't really matter what they
 3131|       |     * point at, as long as the pointers are valid. */
 3132|  45.8k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 3133|  45.8k|    f->lf.p[0] = f->cur.data[0];
 3134|  45.8k|    f->lf.p[1] = f->cur.data[has_chroma ? 1 : 0];
  ------------------
  |  Branch (3134:30): [True: 23.4k, False: 22.4k]
  ------------------
 3135|  45.8k|    f->lf.p[2] = f->cur.data[has_chroma ? 2 : 0];
  ------------------
  |  Branch (3135:30): [True: 23.4k, False: 22.4k]
  ------------------
 3136|  45.8k|    f->lf.sr_p[0] = f->sr_cur.p.data[0];
 3137|  45.8k|    f->lf.sr_p[1] = f->sr_cur.p.data[has_chroma ? 1 : 0];
  ------------------
  |  Branch (3137:38): [True: 23.4k, False: 22.4k]
  ------------------
 3138|  45.8k|    f->lf.sr_p[2] = f->sr_cur.p.data[has_chroma ? 2 : 0];
  ------------------
  |  Branch (3138:38): [True: 23.4k, False: 22.4k]
  ------------------
 3139|       |
 3140|  45.8k|    retval = 0;
 3141|  45.8k|error:
 3142|  45.8k|    return retval;
 3143|  45.8k|}
dav1d_decode_frame_init_cdf:
 3145|  45.8k|int dav1d_decode_frame_init_cdf(Dav1dFrameContext *const f) {
 3146|  45.8k|    const Dav1dContext *const c = f->c;
 3147|  45.8k|    int retval = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  45.8k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3148|       |
 3149|  45.8k|    if (f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3149:9): [True: 15.7k, False: 30.1k]
  ------------------
 3150|  15.7k|        dav1d_cdf_thread_copy(f->out_cdf.data.cdf, &f->in_cdf);
 3151|       |
 3152|       |    // parse individual tiles per tile group
 3153|  45.8k|    int tile_row = 0, tile_col = 0;
 3154|  45.8k|    f->task_thread.update_set = 0;
 3155|  90.3k|    for (int i = 0; i < f->n_tile_data; i++) {
  ------------------
  |  Branch (3155:21): [True: 46.0k, False: 44.2k]
  ------------------
 3156|  46.0k|        const uint8_t *data = f->tile[i].data.data;
 3157|  46.0k|        size_t size = f->tile[i].data.sz;
 3158|       |
 3159|  96.5k|        for (int j = f->tile[i].start; j <= f->tile[i].end; j++) {
  ------------------
  |  Branch (3159:40): [True: 52.0k, False: 44.4k]
  ------------------
 3160|  52.0k|            size_t tile_sz;
 3161|  52.0k|            if (j == f->tile[i].end) {
  ------------------
  |  Branch (3161:17): [True: 44.4k, False: 7.61k]
  ------------------
 3162|  44.4k|                tile_sz = size;
 3163|  44.4k|            } else {
 3164|  7.61k|                if (f->frame_hdr->tiling.n_bytes > size) goto error;
  ------------------
  |  Branch (3164:21): [True: 846, False: 6.76k]
  ------------------
 3165|  6.76k|                tile_sz = 0;
 3166|  14.3k|                for (unsigned k = 0; k < f->frame_hdr->tiling.n_bytes; k++)
  ------------------
  |  Branch (3166:38): [True: 7.53k, False: 6.76k]
  ------------------
 3167|  7.53k|                    tile_sz |= (unsigned)*data++ << (k * 8);
 3168|  6.76k|                tile_sz++;
 3169|  6.76k|                size -= f->frame_hdr->tiling.n_bytes;
 3170|  6.76k|                if (tile_sz > size) goto error;
  ------------------
  |  Branch (3170:21): [True: 732, False: 6.03k]
  ------------------
 3171|  6.76k|            }
 3172|       |
 3173|  50.5k|            setup_tile(&f->ts[j], f, data, tile_sz, tile_row, tile_col++,
 3174|  50.5k|                       c->n_fc > 1 ? f->frame_thread.tile_start_off[j] : 0);
  ------------------
  |  Branch (3174:24): [True: 0, False: 50.5k]
  ------------------
 3175|       |
 3176|  50.5k|            if (tile_col == f->frame_hdr->tiling.cols) {
  ------------------
  |  Branch (3176:17): [True: 45.9k, False: 4.56k]
  ------------------
 3177|  45.9k|                tile_col = 0;
 3178|  45.9k|                tile_row++;
 3179|  45.9k|            }
 3180|  50.5k|            if (j == f->frame_hdr->tiling.update && f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3180:17): [True: 44.4k, False: 6.01k]
  |  Branch (3180:53): [True: 15.4k, False: 29.0k]
  ------------------
 3181|  15.4k|                f->task_thread.update_set = 1;
 3182|  50.5k|            data += tile_sz;
 3183|  50.5k|            size -= tile_sz;
 3184|  50.5k|        }
 3185|  46.0k|    }
 3186|       |
 3187|  44.2k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (3187:9): [True: 0, False: 44.2k]
  ------------------
 3188|      0|        const int uses_2pass = c->n_fc > 1;
 3189|      0|        for (int n = 0; n < f->sb128w * f->frame_hdr->tiling.rows * (1 + uses_2pass); n++)
  ------------------
  |  Branch (3189:25): [True: 0, False: 0]
  ------------------
 3190|      0|            reset_context(&f->a[n], IS_KEY_OR_INTRA(f->frame_hdr),
  ------------------
  |  |   43|      0|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
 3191|      0|                          uses_2pass ? 1 + (n >= f->sb128w * f->frame_hdr->tiling.rows) : 0);
  ------------------
  |  Branch (3191:27): [True: 0, False: 0]
  ------------------
 3192|      0|    }
 3193|       |
 3194|  44.2k|    retval = 0;
 3195|  45.8k|error:
 3196|  45.8k|    return retval;
 3197|  44.2k|}
dav1d_decode_frame_main:
 3199|  44.2k|int dav1d_decode_frame_main(Dav1dFrameContext *const f) {
 3200|  44.2k|    const Dav1dContext *const c = f->c;
 3201|  44.2k|    int retval = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  44.2k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3202|       |
 3203|  44.2k|    assert(f->c->n_tc == 1);
  ------------------
  |  Branch (3203:5): [True: 44.2k, False: 0]
  ------------------
 3204|       |
 3205|  44.2k|    Dav1dTaskContext *const t = &c->tc[f - c->fc];
 3206|  44.2k|    t->f = f;
 3207|  44.2k|    t->frame_thread.pass = 0;
 3208|       |
 3209|   406k|    for (int n = 0; n < f->sb128w * f->frame_hdr->tiling.rows; n++)
  ------------------
  |  Branch (3209:21): [True: 362k, False: 44.2k]
  ------------------
 3210|   362k|        reset_context(&f->a[n], IS_KEY_OR_INTRA(f->frame_hdr), 0);
  ------------------
  |  |   43|   362k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|   362k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
 3211|       |
 3212|       |    // no threading - we explicitly interleave tile/sbrow decoding
 3213|       |    // and post-filtering, so that the full process runs in-line
 3214|  65.5k|    for (int tile_row = 0; tile_row < f->frame_hdr->tiling.rows; tile_row++) {
  ------------------
  |  Branch (3214:28): [True: 45.4k, False: 20.0k]
  ------------------
 3215|  45.4k|        const int sbh_end =
 3216|  45.4k|            imin(f->frame_hdr->tiling.row_start_sb[tile_row + 1], f->sbh);
 3217|  45.4k|        for (int sby = f->frame_hdr->tiling.row_start_sb[tile_row];
 3218|   198k|             sby < sbh_end; sby++)
  ------------------
  |  Branch (3218:14): [True: 176k, False: 21.3k]
  ------------------
 3219|   176k|        {
 3220|   176k|            t->by = sby << (4 + f->seq_hdr->sb128);
 3221|   176k|            const int by_end = (t->by + f->sb_step) >> 1;
 3222|   176k|            if (f->frame_hdr->use_ref_frame_mvs) {
  ------------------
  |  Branch (3222:17): [True: 18.1k, False: 158k]
  ------------------
 3223|  18.1k|                f->c->refmvs_dsp.load_tmvs(&f->rf, tile_row,
 3224|  18.1k|                                           0, f->bw >> 1, t->by >> 1, by_end);
 3225|  18.1k|            }
 3226|   331k|            for (int tile_col = 0; tile_col < f->frame_hdr->tiling.cols; tile_col++) {
  ------------------
  |  Branch (3226:36): [True: 179k, False: 152k]
  ------------------
 3227|   179k|                t->ts = &f->ts[tile_row * f->frame_hdr->tiling.cols + tile_col];
 3228|   179k|                if (dav1d_decode_tile_sbrow(t)) goto error;
  ------------------
  |  Branch (3228:21): [True: 24.1k, False: 155k]
  ------------------
 3229|   179k|            }
 3230|   152k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|   152k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 48.7k, False: 103k]
  |  |  ------------------
  ------------------
 3231|  48.7k|                dav1d_refmvs_save_tmvs(&f->c->refmvs_dsp, &t->rt,
 3232|  48.7k|                                       0, f->bw >> 1, t->by >> 1, by_end);
 3233|  48.7k|            }
 3234|       |
 3235|       |            // loopfilter + cdef + restoration
 3236|   152k|            f->bd_fn.filter_sbrow(f, sby);
 3237|   152k|        }
 3238|  45.4k|    }
 3239|       |
 3240|  20.0k|    retval = 0;
 3241|  44.2k|error:
 3242|  44.2k|    return retval;
 3243|  20.0k|}
dav1d_decode_frame_exit:
 3245|  45.8k|void dav1d_decode_frame_exit(Dav1dFrameContext *const f, int retval) {
 3246|  45.8k|    const Dav1dContext *const c = f->c;
 3247|       |
 3248|  45.8k|    if (f->sr_cur.p.data[0])
  ------------------
  |  Branch (3248:9): [True: 45.8k, False: 0]
  ------------------
 3249|  45.8k|        atomic_init(&f->task_thread.error, 0);
 3250|       |
 3251|  45.8k|    if (c->n_fc > 1 && retval && f->frame_thread.cf) {
  ------------------
  |  Branch (3251:9): [True: 0, False: 45.8k]
  |  Branch (3251:24): [True: 0, False: 0]
  |  Branch (3251:34): [True: 0, False: 0]
  ------------------
 3252|      0|        memset(f->frame_thread.cf, 0,
 3253|      0|               (size_t)f->frame_thread.cf_sz * 128 * 128 / 2);
 3254|      0|    }
 3255|   366k|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3255:21): [True: 320k, False: 45.8k]
  ------------------
 3256|   320k|        if (f->refp[i].p.frame_hdr) {
  ------------------
  |  Branch (3256:13): [True: 94.4k, False: 226k]
  ------------------
 3257|  94.4k|            if (!retval && c->n_fc > 1 && c->strict_std_compliance &&
  ------------------
  |  Branch (3257:17): [True: 63.5k, False: 30.8k]
  |  Branch (3257:28): [True: 0, False: 63.5k]
  |  Branch (3257:43): [True: 0, False: 0]
  ------------------
 3258|  94.4k|                atomic_load(&f->refp[i].progress[1]) == FRAME_ERROR)
  ------------------
  |  |   35|      0|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (3258:17): [True: 0, False: 0]
  ------------------
 3259|      0|            {
 3260|      0|                retval = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3261|      0|                atomic_store(&f->task_thread.error, 1);
 3262|      0|                atomic_store(&f->sr_cur.progress[1], FRAME_ERROR);
 3263|      0|            }
 3264|  94.4k|            dav1d_thread_picture_unref(&f->refp[i]);
 3265|  94.4k|        }
 3266|   320k|        dav1d_ref_dec(&f->ref_mvs_ref[i]);
 3267|   320k|    }
 3268|       |
 3269|  45.8k|    dav1d_picture_unref_internal(&f->cur);
 3270|  45.8k|    dav1d_thread_picture_unref(&f->sr_cur);
 3271|  45.8k|    dav1d_cdf_thread_unref(&f->in_cdf);
 3272|  45.8k|    if (f->frame_hdr && f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3272:9): [True: 45.8k, False: 0]
  |  Branch (3272:25): [True: 15.7k, False: 30.1k]
  ------------------
 3273|  15.7k|        if (f->out_cdf.progress)
  ------------------
  |  Branch (3273:13): [True: 0, False: 15.7k]
  ------------------
 3274|  15.7k|            atomic_store(f->out_cdf.progress, retval == 0 ? 1 : TILE_ERROR);
  ------------------
  |  Branch (3274:13): [True: 0, False: 0]
  ------------------
 3275|  15.7k|        dav1d_cdf_thread_unref(&f->out_cdf);
 3276|  15.7k|    }
 3277|  45.8k|    dav1d_ref_dec(&f->cur_segmap_ref);
 3278|  45.8k|    dav1d_ref_dec(&f->prev_segmap_ref);
 3279|  45.8k|    dav1d_ref_dec(&f->mvs_ref);
 3280|  45.8k|    dav1d_ref_dec(&f->seq_hdr_ref);
 3281|  45.8k|    dav1d_ref_dec(&f->frame_hdr_ref);
 3282|       |
 3283|  91.8k|    for (int i = 0; i < f->n_tile_data; i++)
  ------------------
  |  Branch (3283:21): [True: 46.0k, False: 45.8k]
  ------------------
 3284|  46.0k|        dav1d_data_unref_internal(&f->tile[i].data);
 3285|  45.8k|    f->task_thread.retval = retval;
 3286|  45.8k|}
dav1d_decode_frame:
 3288|  45.8k|int dav1d_decode_frame(Dav1dFrameContext *const f) {
 3289|  45.8k|    assert(f->c->n_fc == 1);
  ------------------
  |  Branch (3289:5): [True: 45.8k, False: 0]
  ------------------
 3290|       |    // if n_tc > 1 (but n_fc == 1), we could run init/exit in the task
 3291|       |    // threads also. Not sure it makes a measurable difference.
 3292|  45.8k|    int res = dav1d_decode_frame_init(f);
 3293|  45.8k|    if (!res) res = dav1d_decode_frame_init_cdf(f);
  ------------------
  |  Branch (3293:9): [True: 45.8k, False: 0]
  ------------------
 3294|       |    // wait until all threads have completed
 3295|  45.8k|    if (!res) {
  ------------------
  |  Branch (3295:9): [True: 44.2k, False: 1.57k]
  ------------------
 3296|  44.2k|        if (f->c->n_tc > 1) {
  ------------------
  |  Branch (3296:13): [True: 0, False: 44.2k]
  ------------------
 3297|      0|            res = dav1d_task_create_tile_sbrow(f, 0, 1);
 3298|      0|            pthread_mutex_lock(&f->task_thread.ttd->lock);
 3299|      0|            pthread_cond_signal(&f->task_thread.ttd->cond);
 3300|      0|            if (!res) {
  ------------------
  |  Branch (3300:17): [True: 0, False: 0]
  ------------------
 3301|      0|                while (!f->task_thread.done[0] ||
  ------------------
  |  Branch (3301:24): [True: 0, False: 0]
  ------------------
 3302|      0|                       atomic_load(&f->task_thread.task_counter) > 0)
  ------------------
  |  Branch (3302:24): [True: 0, False: 0]
  ------------------
 3303|      0|                {
 3304|      0|                    pthread_cond_wait(&f->task_thread.cond,
 3305|      0|                                      &f->task_thread.ttd->lock);
 3306|      0|                }
 3307|      0|            }
 3308|      0|            pthread_mutex_unlock(&f->task_thread.ttd->lock);
 3309|      0|            res = f->task_thread.retval;
 3310|  44.2k|        } else {
 3311|  44.2k|            res = dav1d_decode_frame_main(f);
 3312|  44.2k|            if (!res && f->frame_hdr->refresh_context && f->task_thread.update_set) {
  ------------------
  |  Branch (3312:17): [True: 20.0k, False: 24.1k]
  |  Branch (3312:25): [True: 13.4k, False: 6.64k]
  |  Branch (3312:58): [True: 13.4k, False: 0]
  ------------------
 3313|  13.4k|                dav1d_cdf_thread_update(f->frame_hdr, f->out_cdf.data.cdf,
 3314|  13.4k|                                        &f->ts[f->frame_hdr->tiling.update].cdf);
 3315|  13.4k|            }
 3316|  44.2k|        }
 3317|  44.2k|    }
 3318|  45.8k|    dav1d_decode_frame_exit(f, res);
 3319|  45.8k|    res = f->task_thread.retval;
 3320|  45.8k|    f->n_tile_data = 0;
 3321|  45.8k|    return res;
 3322|  45.8k|}
dav1d_submit_frame:
 3330|  48.4k|int dav1d_submit_frame(Dav1dContext *const c) {
 3331|  48.4k|    Dav1dFrameContext *f;
 3332|  48.4k|    int res = -1;
 3333|       |
 3334|       |    // wait for c->out_delayed[next] and move into c->out if visible
 3335|  48.4k|    Dav1dThreadPicture *out_delayed;
 3336|  48.4k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (3336:9): [True: 0, False: 48.4k]
  ------------------
 3337|      0|        pthread_mutex_lock(&c->task_thread.lock);
 3338|      0|        const unsigned next = c->frame_thread.next++;
 3339|      0|        if (c->frame_thread.next == c->n_fc)
  ------------------
  |  Branch (3339:13): [True: 0, False: 0]
  ------------------
 3340|      0|            c->frame_thread.next = 0;
 3341|       |
 3342|      0|        f = &c->fc[next];
 3343|      0|        while (f->n_tile_data > 0)
  ------------------
  |  Branch (3343:16): [True: 0, False: 0]
  ------------------
 3344|      0|            pthread_cond_wait(&f->task_thread.cond,
 3345|      0|                              &c->task_thread.lock);
 3346|      0|        out_delayed = &c->frame_thread.out_delayed[next];
 3347|      0|        if (out_delayed->p.data[0] || atomic_load(&f->task_thread.error)) {
  ------------------
  |  Branch (3347:13): [True: 0, False: 0]
  |  Branch (3347:39): [True: 0, False: 0]
  ------------------
 3348|      0|            unsigned first = atomic_load(&c->task_thread.first);
 3349|      0|            if (first + 1U < c->n_fc)
  ------------------
  |  Branch (3349:17): [True: 0, False: 0]
  ------------------
 3350|      0|                atomic_fetch_add(&c->task_thread.first, 1U);
 3351|      0|            else
 3352|      0|                atomic_store(&c->task_thread.first, 0);
 3353|      0|            atomic_compare_exchange_strong(&c->task_thread.reset_task_cur,
 3354|      0|                                           &first, UINT_MAX);
 3355|      0|            if (c->task_thread.cur && c->task_thread.cur < c->n_fc)
  ------------------
  |  Branch (3355:17): [True: 0, False: 0]
  |  Branch (3355:39): [True: 0, False: 0]
  ------------------
 3356|      0|                c->task_thread.cur--;
 3357|      0|        }
 3358|      0|        const int error = f->task_thread.retval;
 3359|      0|        if (error) {
  ------------------
  |  Branch (3359:13): [True: 0, False: 0]
  ------------------
 3360|      0|            f->task_thread.retval = 0;
 3361|      0|            c->cached_error = error;
 3362|      0|            dav1d_data_props_copy(&c->cached_error_props, &out_delayed->p.m);
 3363|      0|            dav1d_thread_picture_unref(out_delayed);
 3364|      0|        } else if (out_delayed->p.data[0]) {
  ------------------
  |  Branch (3364:20): [True: 0, False: 0]
  ------------------
 3365|      0|            const unsigned progress = atomic_load_explicit(&out_delayed->progress[1],
 3366|      0|                                                           memory_order_relaxed);
 3367|      0|            if ((out_delayed->visible || c->output_invisible_frames) &&
  ------------------
  |  Branch (3367:18): [True: 0, False: 0]
  |  Branch (3367:42): [True: 0, False: 0]
  ------------------
 3368|      0|                progress != FRAME_ERROR)
  ------------------
  |  |   35|      0|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (3368:17): [True: 0, False: 0]
  ------------------
 3369|      0|            {
 3370|      0|                dav1d_thread_picture_ref(&c->out, out_delayed);
 3371|      0|                c->event_flags |= dav1d_picture_get_event_flags(out_delayed);
 3372|      0|            }
 3373|      0|            dav1d_thread_picture_unref(out_delayed);
 3374|      0|        }
 3375|  48.4k|    } else {
 3376|  48.4k|        f = c->fc;
 3377|  48.4k|    }
 3378|       |
 3379|  48.4k|    f->seq_hdr = c->seq_hdr;
 3380|  48.4k|    f->seq_hdr_ref = c->seq_hdr_ref;
 3381|  48.4k|    dav1d_ref_inc(f->seq_hdr_ref);
 3382|  48.4k|    f->frame_hdr = c->frame_hdr;
 3383|  48.4k|    f->frame_hdr_ref = c->frame_hdr_ref;
 3384|  48.4k|    c->frame_hdr = NULL;
 3385|  48.4k|    c->frame_hdr_ref = NULL;
 3386|  48.4k|    f->dsp = &c->dsp[f->seq_hdr->hbd];
 3387|       |
 3388|  48.4k|    const int bpc = 8 + 2 * f->seq_hdr->hbd;
 3389|       |
 3390|  48.4k|    if (!f->dsp->ipred.intra_pred[DC_PRED]) {
  ------------------
  |  Branch (3390:9): [True: 8.66k, False: 39.8k]
  ------------------
 3391|  8.66k|        Dav1dDSPContext *const dsp = &c->dsp[f->seq_hdr->hbd];
 3392|       |
 3393|  8.66k|        switch (bpc) {
 3394|      0|#define assign_bitdepth_case(bd) \
 3395|      0|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
 3396|      0|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
 3397|      0|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
 3398|      0|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
 3399|      0|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
 3400|      0|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
 3401|      0|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
 3402|      0|            break
 3403|      0|#if CONFIG_8BPC
 3404|  3.66k|        case 8:
  ------------------
  |  Branch (3404:9): [True: 3.66k, False: 4.99k]
  ------------------
 3405|  3.66k|            assign_bitdepth_case(8);
  ------------------
  |  | 3395|  3.66k|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
  |  | 3396|  3.66k|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
  |  | 3397|  3.66k|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
  |  | 3398|  3.66k|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
  |  | 3399|  3.66k|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
  |  | 3400|  3.66k|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
  |  | 3401|  3.66k|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
  |  | 3402|  3.66k|            break
  ------------------
 3406|      0|#endif
 3407|      0|#if CONFIG_16BPC
 3408|  2.47k|        case 10:
  ------------------
  |  Branch (3408:9): [True: 2.47k, False: 6.18k]
  ------------------
 3409|  4.99k|        case 12:
  ------------------
  |  Branch (3409:9): [True: 2.52k, False: 6.14k]
  ------------------
 3410|  4.99k|            assign_bitdepth_case(16);
  ------------------
  |  | 3395|  4.99k|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
  |  | 3396|  4.99k|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
  |  | 3397|  4.99k|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
  |  | 3398|  4.99k|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
  |  | 3399|  4.99k|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
  |  | 3400|  4.99k|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
  |  | 3401|  4.99k|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
  |  | 3402|  4.99k|            break
  ------------------
 3411|      0|#endif
 3412|      0|#undef assign_bitdepth_case
 3413|      0|        default:
  ------------------
  |  Branch (3413:9): [True: 0, False: 8.66k]
  ------------------
 3414|      0|            dav1d_log(c, "Compiled without support for %d-bit decoding\n",
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
 3415|      0|                    8 + 2 * f->seq_hdr->hbd);
 3416|      0|            res = DAV1D_ERR(ENOPROTOOPT);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3417|      0|            goto error;
 3418|  8.66k|        }
 3419|  8.66k|    }
 3420|       |
 3421|  48.4k|#define assign_bitdepth_case(bd) \
 3422|  48.4k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
 3423|  48.4k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
 3424|  48.4k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
 3425|  48.4k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
 3426|  48.4k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
 3427|  48.4k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
 3428|  48.4k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
 3429|  48.4k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
 3430|  48.4k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
 3431|  48.4k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
 3432|  48.4k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
 3433|  48.4k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
 3434|  48.4k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
 3435|  48.4k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
 3436|  48.4k|    if (!f->seq_hdr->hbd) {
  ------------------
  |  Branch (3436:9): [True: 25.2k, False: 23.2k]
  ------------------
 3437|  25.2k|#if CONFIG_8BPC
 3438|  25.2k|        assign_bitdepth_case(8);
  ------------------
  |  | 3422|  25.2k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
  |  | 3423|  25.2k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
  |  | 3424|  25.2k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
  |  | 3425|  25.2k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
  |  | 3426|  25.2k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
  |  | 3427|  25.2k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
  |  | 3428|  25.2k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
  |  | 3429|  25.2k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
  |  | 3430|  25.2k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
  |  | 3431|  25.2k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
  |  | 3432|  25.2k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
  |  | 3433|  25.2k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
  |  | 3434|  25.2k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
  |  | 3435|  25.2k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
  ------------------
 3439|  25.2k|#endif
 3440|  25.2k|    } else {
 3441|  23.2k|#if CONFIG_16BPC
 3442|  23.2k|        assign_bitdepth_case(16);
  ------------------
  |  | 3422|  23.2k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
  |  | 3423|  23.2k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
  |  | 3424|  23.2k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
  |  | 3425|  23.2k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
  |  | 3426|  23.2k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
  |  | 3427|  23.2k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
  |  | 3428|  23.2k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
  |  | 3429|  23.2k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
  |  | 3430|  23.2k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
  |  | 3431|  23.2k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
  |  | 3432|  23.2k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
  |  | 3433|  23.2k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
  |  | 3434|  23.2k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
  |  | 3435|  23.2k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
  ------------------
 3443|  23.2k|#endif
 3444|  23.2k|    }
 3445|  48.4k|#undef assign_bitdepth_case
 3446|       |
 3447|  48.4k|    int ref_coded_width[7];
 3448|  48.4k|    if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  48.4k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 16.1k, False: 32.3k]
  |  |  ------------------
  ------------------
 3449|  16.1k|        if (f->frame_hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  16.1k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (3449:13): [True: 14.2k, False: 1.90k]
  ------------------
 3450|  14.2k|            const int pri_ref = f->frame_hdr->refidx[f->frame_hdr->primary_ref_frame];
 3451|  14.2k|            if (!c->refs[pri_ref].p.p.data[0]) {
  ------------------
  |  Branch (3451:17): [True: 231, False: 13.9k]
  ------------------
 3452|    231|                res = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    231|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3453|    231|                goto error;
 3454|    231|            }
 3455|  14.2k|        }
 3456|   114k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3456:25): [True: 101k, False: 13.4k]
  ------------------
 3457|   101k|            const int refidx = f->frame_hdr->refidx[i];
 3458|   101k|            if (!c->refs[refidx].p.p.data[0] ||
  ------------------
  |  Branch (3458:17): [True: 479, False: 100k]
  ------------------
 3459|   100k|                f->frame_hdr->width[0] * 2 < c->refs[refidx].p.p.p.w ||
  ------------------
  |  Branch (3459:17): [True: 368, False: 100k]
  ------------------
 3460|   100k|                f->frame_hdr->height * 2 < c->refs[refidx].p.p.p.h ||
  ------------------
  |  Branch (3460:17): [True: 233, False: 100k]
  ------------------
 3461|   100k|                f->frame_hdr->width[0] > c->refs[refidx].p.p.p.w * 16 ||
  ------------------
  |  Branch (3461:17): [True: 769, False: 99.2k]
  ------------------
 3462|  99.2k|                f->frame_hdr->height > c->refs[refidx].p.p.p.h * 16 ||
  ------------------
  |  Branch (3462:17): [True: 548, False: 98.6k]
  ------------------
 3463|  98.6k|                f->seq_hdr->layout != c->refs[refidx].p.p.p.layout ||
  ------------------
  |  Branch (3463:17): [True: 0, False: 98.6k]
  ------------------
 3464|  98.6k|                bpc != c->refs[refidx].p.p.p.bpc)
  ------------------
  |  Branch (3464:17): [True: 0, False: 98.6k]
  ------------------
 3465|  2.39k|            {
 3466|  6.66k|                for (int j = 0; j < i; j++)
  ------------------
  |  Branch (3466:33): [True: 4.27k, False: 2.39k]
  ------------------
 3467|  4.27k|                    dav1d_thread_picture_unref(&f->refp[j]);
 3468|  2.39k|                res = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  2.39k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3469|  2.39k|                goto error;
 3470|  2.39k|            }
 3471|  98.6k|            dav1d_thread_picture_ref(&f->refp[i], &c->refs[refidx].p);
 3472|  98.6k|            ref_coded_width[i] = c->refs[refidx].p.p.frame_hdr->width[0];
 3473|  98.6k|            if (f->frame_hdr->width[0] != c->refs[refidx].p.p.p.w ||
  ------------------
  |  Branch (3473:17): [True: 17.7k, False: 80.9k]
  ------------------
 3474|  80.9k|                f->frame_hdr->height != c->refs[refidx].p.p.p.h)
  ------------------
  |  Branch (3474:17): [True: 4.80k, False: 76.1k]
  ------------------
 3475|  22.5k|            {
 3476|  22.5k|#define scale_fac(ref_sz, this_sz) \
 3477|  22.5k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
 3478|  22.5k|                f->svc[i][0].scale = scale_fac(c->refs[refidx].p.p.p.w,
  ------------------
  |  | 3477|  22.5k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3479|  22.5k|                                               f->frame_hdr->width[0]);
 3480|  22.5k|                f->svc[i][1].scale = scale_fac(c->refs[refidx].p.p.p.h,
  ------------------
  |  | 3477|  22.5k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3481|  22.5k|                                               f->frame_hdr->height);
 3482|  22.5k|                f->svc[i][0].step = (f->svc[i][0].scale + 8) >> 4;
 3483|  22.5k|                f->svc[i][1].step = (f->svc[i][1].scale + 8) >> 4;
 3484|  76.1k|            } else {
 3485|  76.1k|                f->svc[i][0].scale = f->svc[i][1].scale = 0;
 3486|  76.1k|            }
 3487|  98.6k|            f->gmv_warp_allowed[i] = f->frame_hdr->gmv[i].type > DAV1D_WM_TYPE_TRANSLATION &&
  ------------------
  |  Branch (3487:38): [True: 5.42k, False: 93.2k]
  ------------------
 3488|  5.42k|                                     !f->frame_hdr->force_integer_mv &&
  ------------------
  |  Branch (3488:38): [True: 4.62k, False: 801]
  ------------------
 3489|  4.62k|                                     !dav1d_get_shear_params(&f->frame_hdr->gmv[i]) &&
  ------------------
  |  Branch (3489:38): [True: 3.66k, False: 962]
  ------------------
 3490|  3.66k|                                     !f->svc[i][0].scale;
  ------------------
  |  Branch (3490:38): [True: 2.56k, False: 1.10k]
  ------------------
 3491|  98.6k|        }
 3492|  15.8k|    }
 3493|       |
 3494|       |    // setup entropy
 3495|  45.8k|    if (f->frame_hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  45.8k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (3495:9): [True: 33.7k, False: 12.0k]
  ------------------
 3496|  33.7k|        dav1d_cdf_thread_init_static(&f->in_cdf, f->frame_hdr->quant.yac);
 3497|  33.7k|    } else {
 3498|  12.0k|        const int pri_ref = f->frame_hdr->refidx[f->frame_hdr->primary_ref_frame];
 3499|  12.0k|        dav1d_cdf_thread_ref(&f->in_cdf, &c->cdf[pri_ref]);
 3500|  12.0k|    }
 3501|  45.8k|    if (f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3501:9): [True: 15.7k, False: 30.1k]
  ------------------
 3502|  15.7k|        res = dav1d_cdf_thread_alloc(c, &f->out_cdf, c->n_fc > 1);
 3503|  15.7k|        if (res < 0) goto error;
  ------------------
  |  Branch (3503:13): [True: 0, False: 15.7k]
  ------------------
 3504|  15.7k|    }
 3505|       |
 3506|       |    // FIXME qsort so tiles are in order (for frame threading)
 3507|  45.8k|    if (f->n_tile_data_alloc < c->n_tile_data) {
  ------------------
  |  Branch (3507:9): [True: 8.58k, False: 37.2k]
  ------------------
 3508|  8.58k|        dav1d_free(f->tile);
  ------------------
  |  |  135|  8.58k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3509|  8.58k|        assert(c->n_tile_data < INT_MAX / (int)sizeof(*f->tile));
  ------------------
  |  Branch (3509:9): [True: 8.58k, False: 0]
  ------------------
 3510|  8.58k|        f->tile = dav1d_malloc(ALLOC_TILE, c->n_tile_data * sizeof(*f->tile));
  ------------------
  |  |  132|  8.58k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3511|  8.58k|        if (!f->tile) {
  ------------------
  |  Branch (3511:13): [True: 0, False: 8.58k]
  ------------------
 3512|      0|            f->n_tile_data_alloc = f->n_tile_data = 0;
 3513|      0|            res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3514|      0|            goto error;
 3515|      0|        }
 3516|  8.58k|        f->n_tile_data_alloc = c->n_tile_data;
 3517|  8.58k|    }
 3518|  45.8k|    memcpy(f->tile, c->tile, c->n_tile_data * sizeof(*f->tile));
 3519|  45.8k|    memset(c->tile, 0, c->n_tile_data * sizeof(*c->tile));
 3520|  45.8k|    f->n_tile_data = c->n_tile_data;
 3521|  45.8k|    c->n_tile_data = 0;
 3522|       |
 3523|       |    // allocate frame
 3524|  45.8k|    res = dav1d_thread_picture_alloc(c, f, bpc);
 3525|  45.8k|    if (res < 0) goto error;
  ------------------
  |  Branch (3525:9): [True: 0, False: 45.8k]
  ------------------
 3526|       |
 3527|  45.8k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (3527:9): [True: 4.83k, False: 41.0k]
  ------------------
 3528|  4.83k|        res = dav1d_picture_alloc_copy(c, &f->cur, f->frame_hdr->width[0], &f->sr_cur.p);
 3529|  4.83k|        if (res < 0) goto error;
  ------------------
  |  Branch (3529:13): [True: 0, False: 4.83k]
  ------------------
 3530|  41.0k|    } else {
 3531|  41.0k|        dav1d_picture_ref(&f->cur, &f->sr_cur.p);
 3532|  41.0k|    }
 3533|       |
 3534|  45.8k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (3534:9): [True: 4.83k, False: 41.0k]
  ------------------
 3535|  4.83k|        f->resize_step[0] = scale_fac(f->cur.p.w, f->sr_cur.p.p.w);
  ------------------
  |  | 3477|  4.83k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3536|  4.83k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 3537|  4.83k|        const int in_cw = (f->cur.p.w + ss_hor) >> ss_hor;
 3538|  4.83k|        const int out_cw = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 3539|  4.83k|        f->resize_step[1] = scale_fac(in_cw, out_cw);
  ------------------
  |  | 3477|  4.83k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3540|  4.83k|#undef scale_fac
 3541|  4.83k|        f->resize_start[0] = get_upscale_x0(f->cur.p.w, f->sr_cur.p.p.w, f->resize_step[0]);
 3542|  4.83k|        f->resize_start[1] = get_upscale_x0(in_cw, out_cw, f->resize_step[1]);
 3543|  4.83k|    }
 3544|       |
 3545|       |    // move f->cur into output queue
 3546|  45.8k|    if (c->n_fc == 1) {
  ------------------
  |  Branch (3546:9): [True: 45.8k, False: 0]
  ------------------
 3547|  45.8k|        if (f->frame_hdr->show_frame || c->output_invisible_frames) {
  ------------------
  |  Branch (3547:13): [True: 41.3k, False: 4.47k]
  |  Branch (3547:41): [True: 0, False: 4.47k]
  ------------------
 3548|  41.3k|            dav1d_thread_picture_ref(&c->out, &f->sr_cur);
 3549|  41.3k|            c->event_flags |= dav1d_picture_get_event_flags(&f->sr_cur);
 3550|  41.3k|        }
 3551|  45.8k|    } else {
 3552|      0|        dav1d_thread_picture_ref(out_delayed, &f->sr_cur);
 3553|      0|    }
 3554|       |
 3555|  45.8k|    f->w4 = (f->frame_hdr->width[0] + 3) >> 2;
 3556|  45.8k|    f->h4 = (f->frame_hdr->height + 3) >> 2;
 3557|  45.8k|    f->bw = ((f->frame_hdr->width[0] + 7) >> 3) << 1;
 3558|  45.8k|    f->bh = ((f->frame_hdr->height + 7) >> 3) << 1;
 3559|  45.8k|    f->sb128w = (f->bw + 31) >> 5;
 3560|  45.8k|    f->sb128h = (f->bh + 31) >> 5;
 3561|  45.8k|    f->sb_shift = 4 + f->seq_hdr->sb128;
 3562|  45.8k|    f->sb_step = 16 << f->seq_hdr->sb128;
 3563|  45.8k|    f->sbh = (f->bh + f->sb_step - 1) >> f->sb_shift;
 3564|  45.8k|    f->b4_stride = (f->bw + 31) & ~31;
 3565|  45.8k|    f->bitdepth_max = (1 << f->cur.p.bpc) - 1;
 3566|  45.8k|    atomic_init(&f->task_thread.error, 0);
 3567|  45.8k|    const int uses_2pass = c->n_fc > 1;
 3568|  45.8k|    const int cols = f->frame_hdr->tiling.cols;
 3569|  45.8k|    const int rows = f->frame_hdr->tiling.rows;
 3570|  45.8k|    atomic_store(&f->task_thread.task_counter,
 3571|  45.8k|                 (cols * rows + f->sbh) << uses_2pass);
 3572|       |
 3573|       |    // ref_mvs
 3574|  45.8k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|  91.6k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 13.4k, False: 32.3k]
  |  |  ------------------
  ------------------
  |  Branch (3574:45): [True: 17.8k, False: 14.4k]
  ------------------
 3575|  31.3k|        f->mvs_ref = dav1d_ref_create_using_pool(c->refmvs_pool,
 3576|  31.3k|            sizeof(*f->mvs) * f->sb128h * 16 * (f->b4_stride >> 1));
 3577|  31.3k|        if (!f->mvs_ref) {
  ------------------
  |  Branch (3577:13): [True: 0, False: 31.3k]
  ------------------
 3578|      0|            res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3579|      0|            goto error;
 3580|      0|        }
 3581|  31.3k|        f->mvs = f->mvs_ref->data;
 3582|  31.3k|        if (!f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (3582:13): [True: 13.4k, False: 17.8k]
  ------------------
 3583|   107k|            for (int i = 0; i < 7; i++)
  ------------------
  |  Branch (3583:29): [True: 94.4k, False: 13.4k]
  ------------------
 3584|  94.4k|                f->refpoc[i] = f->refp[i].p.frame_hdr->frame_offset;
 3585|  17.8k|        } else {
 3586|  17.8k|            memset(f->refpoc, 0, sizeof(f->refpoc));
 3587|  17.8k|        }
 3588|  31.3k|        if (f->frame_hdr->use_ref_frame_mvs) {
  ------------------
  |  Branch (3588:13): [True: 9.45k, False: 21.9k]
  ------------------
 3589|  75.6k|            for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3589:29): [True: 66.1k, False: 9.45k]
  ------------------
 3590|  66.1k|                const int refidx = f->frame_hdr->refidx[i];
 3591|  66.1k|                const int ref_w = ((ref_coded_width[i] + 7) >> 3) << 1;
 3592|  66.1k|                const int ref_h = ((f->refp[i].p.p.h + 7) >> 3) << 1;
 3593|  66.1k|                if (c->refs[refidx].refmvs != NULL &&
  ------------------
  |  Branch (3593:21): [True: 43.7k, False: 22.4k]
  ------------------
 3594|  43.7k|                    ref_w == f->bw && ref_h == f->bh)
  ------------------
  |  Branch (3594:21): [True: 41.1k, False: 2.57k]
  |  Branch (3594:39): [True: 40.8k, False: 303]
  ------------------
 3595|  40.8k|                {
 3596|  40.8k|                    f->ref_mvs_ref[i] = c->refs[refidx].refmvs;
 3597|  40.8k|                    dav1d_ref_inc(f->ref_mvs_ref[i]);
 3598|  40.8k|                    f->ref_mvs[i] = c->refs[refidx].refmvs->data;
 3599|  40.8k|                } else {
 3600|  25.3k|                    f->ref_mvs[i] = NULL;
 3601|  25.3k|                    f->ref_mvs_ref[i] = NULL;
 3602|  25.3k|                }
 3603|  66.1k|                memcpy(f->refrefpoc[i], c->refs[refidx].refpoc,
 3604|  66.1k|                       sizeof(*f->refrefpoc));
 3605|  66.1k|            }
 3606|  21.9k|        } else {
 3607|  21.9k|            memset(f->ref_mvs_ref, 0, sizeof(f->ref_mvs_ref));
 3608|  21.9k|        }
 3609|  31.3k|    } else {
 3610|  14.4k|        f->mvs_ref = NULL;
 3611|  14.4k|        memset(f->ref_mvs_ref, 0, sizeof(f->ref_mvs_ref));
 3612|  14.4k|    }
 3613|       |
 3614|       |    // segmap
 3615|  45.8k|    if (f->frame_hdr->segmentation.enabled) {
  ------------------
  |  Branch (3615:9): [True: 9.45k, False: 36.3k]
  ------------------
 3616|       |        // By default, the previous segmentation map is not initialised.
 3617|  9.45k|        f->prev_segmap_ref = NULL;
 3618|  9.45k|        f->prev_segmap = NULL;
 3619|       |
 3620|       |        // We might need a previous frame's segmentation map. This
 3621|       |        // happens if there is either no update or a temporal update.
 3622|  9.45k|        if (f->frame_hdr->segmentation.temporal || !f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (3622:13): [True: 581, False: 8.87k]
  |  Branch (3622:52): [True: 6.60k, False: 2.26k]
  ------------------
 3623|  7.18k|            const int pri_ref = f->frame_hdr->primary_ref_frame;
 3624|  7.18k|            assert(pri_ref != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (3624:13): [True: 7.18k, False: 0]
  ------------------
 3625|  7.18k|            const int ref_w = ((ref_coded_width[pri_ref] + 7) >> 3) << 1;
 3626|  7.18k|            const int ref_h = ((f->refp[pri_ref].p.p.h + 7) >> 3) << 1;
 3627|  7.18k|            if (ref_w == f->bw && ref_h == f->bh) {
  ------------------
  |  Branch (3627:17): [True: 6.09k, False: 1.09k]
  |  Branch (3627:35): [True: 5.85k, False: 242]
  ------------------
 3628|  5.85k|                f->prev_segmap_ref = c->refs[f->frame_hdr->refidx[pri_ref]].segmap;
 3629|  5.85k|                if (f->prev_segmap_ref) {
  ------------------
  |  Branch (3629:21): [True: 5.40k, False: 452]
  ------------------
 3630|  5.40k|                    dav1d_ref_inc(f->prev_segmap_ref);
 3631|  5.40k|                    f->prev_segmap = f->prev_segmap_ref->data;
 3632|  5.40k|                }
 3633|  5.85k|            }
 3634|  7.18k|        }
 3635|       |
 3636|  9.45k|        if (f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (3636:13): [True: 2.84k, False: 6.60k]
  ------------------
 3637|       |            // We're updating an existing map, but need somewhere to
 3638|       |            // put the new values. Allocate them here (the data
 3639|       |            // actually gets set elsewhere)
 3640|  2.84k|            f->cur_segmap_ref = dav1d_ref_create_using_pool(c->segmap_pool,
 3641|  2.84k|                sizeof(*f->cur_segmap) * f->b4_stride * 32 * f->sb128h);
 3642|  2.84k|            if (!f->cur_segmap_ref) {
  ------------------
  |  Branch (3642:17): [True: 0, False: 2.84k]
  ------------------
 3643|      0|                dav1d_ref_dec(&f->prev_segmap_ref);
 3644|      0|                res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3645|      0|                goto error;
 3646|      0|            }
 3647|  2.84k|            f->cur_segmap = f->cur_segmap_ref->data;
 3648|  6.60k|        } else if (f->prev_segmap_ref) {
  ------------------
  |  Branch (3648:20): [True: 5.04k, False: 1.56k]
  ------------------
 3649|       |            // We're not updating an existing map, and we have a valid
 3650|       |            // reference. Use that.
 3651|  5.04k|            f->cur_segmap_ref = f->prev_segmap_ref;
 3652|  5.04k|            dav1d_ref_inc(f->cur_segmap_ref);
 3653|  5.04k|            f->cur_segmap = f->prev_segmap_ref->data;
 3654|  5.04k|        } else {
 3655|       |            // We need to make a new map. Allocate one here and zero it out.
 3656|  1.56k|            const size_t segmap_size = sizeof(*f->cur_segmap) * f->b4_stride * 32 * f->sb128h;
 3657|  1.56k|            f->cur_segmap_ref = dav1d_ref_create_using_pool(c->segmap_pool, segmap_size);
 3658|  1.56k|            if (!f->cur_segmap_ref) {
  ------------------
  |  Branch (3658:17): [True: 0, False: 1.56k]
  ------------------
 3659|      0|                res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3660|      0|                goto error;
 3661|      0|            }
 3662|  1.56k|            f->cur_segmap = f->cur_segmap_ref->data;
 3663|  1.56k|            memset(f->cur_segmap, 0, segmap_size);
 3664|  1.56k|        }
 3665|  36.3k|    } else {
 3666|  36.3k|        f->cur_segmap = NULL;
 3667|  36.3k|        f->cur_segmap_ref = NULL;
 3668|  36.3k|        f->prev_segmap_ref = NULL;
 3669|  36.3k|    }
 3670|       |
 3671|       |    // update references etc.
 3672|  45.8k|    const unsigned refresh_frame_flags = f->frame_hdr->refresh_frame_flags;
 3673|   412k|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (3673:21): [True: 366k, False: 45.8k]
  ------------------
 3674|   366k|        if (refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (3674:13): [True: 299k, False: 67.1k]
  ------------------
 3675|   299k|            if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (3675:17): [True: 97.2k, False: 202k]
  ------------------
 3676|  97.2k|                dav1d_thread_picture_unref(&c->refs[i].p);
 3677|   299k|            dav1d_thread_picture_ref(&c->refs[i].p, &f->sr_cur);
 3678|       |
 3679|   299k|            dav1d_cdf_thread_unref(&c->cdf[i]);
 3680|   299k|            if (f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3680:17): [True: 82.0k, False: 217k]
  ------------------
 3681|  82.0k|                dav1d_cdf_thread_ref(&c->cdf[i], &f->out_cdf);
 3682|   217k|            } else {
 3683|   217k|                dav1d_cdf_thread_ref(&c->cdf[i], &f->in_cdf);
 3684|   217k|            }
 3685|       |
 3686|   299k|            dav1d_ref_dec(&c->refs[i].segmap);
 3687|   299k|            c->refs[i].segmap = f->cur_segmap_ref;
 3688|   299k|            if (f->cur_segmap_ref)
  ------------------
  |  Branch (3688:17): [True: 51.9k, False: 247k]
  ------------------
 3689|  51.9k|                dav1d_ref_inc(f->cur_segmap_ref);
 3690|   299k|            dav1d_ref_dec(&c->refs[i].refmvs);
 3691|   299k|            if (!f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (3691:17): [True: 157k, False: 142k]
  ------------------
 3692|   157k|                c->refs[i].refmvs = f->mvs_ref;
 3693|   157k|                if (f->mvs_ref)
  ------------------
  |  Branch (3693:21): [True: 47.2k, False: 109k]
  ------------------
 3694|  47.2k|                    dav1d_ref_inc(f->mvs_ref);
 3695|   157k|            }
 3696|   299k|            memcpy(c->refs[i].refpoc, f->refpoc, sizeof(f->refpoc));
 3697|   299k|        }
 3698|   366k|    }
 3699|       |
 3700|  45.8k|    if (c->n_fc == 1) {
  ------------------
  |  Branch (3700:9): [True: 45.8k, False: 0]
  ------------------
 3701|  45.8k|        if ((res = dav1d_decode_frame(f)) < 0) {
  ------------------
  |  Branch (3701:13): [True: 25.7k, False: 20.0k]
  ------------------
 3702|  25.7k|            dav1d_thread_picture_unref(&c->out);
 3703|   231k|            for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (3703:29): [True: 206k, False: 25.7k]
  ------------------
 3704|   206k|                if (refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (3704:21): [True: 179k, False: 26.9k]
  ------------------
 3705|   179k|                    if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (3705:25): [True: 179k, False: 0]
  ------------------
 3706|   179k|                        dav1d_thread_picture_unref(&c->refs[i].p);
 3707|   179k|                    dav1d_cdf_thread_unref(&c->cdf[i]);
 3708|   179k|                    dav1d_ref_dec(&c->refs[i].segmap);
 3709|   179k|                    dav1d_ref_dec(&c->refs[i].refmvs);
 3710|   179k|                }
 3711|   206k|            }
 3712|  25.7k|            goto error;
 3713|  25.7k|        }
 3714|  45.8k|    } else {
 3715|      0|        dav1d_task_frame_init(f);
 3716|      0|        pthread_mutex_unlock(&c->task_thread.lock);
 3717|      0|    }
 3718|       |
 3719|  20.0k|    return 0;
 3720|  28.3k|error:
 3721|  28.3k|    atomic_init(&f->task_thread.error, 1);
 3722|  28.3k|    dav1d_cdf_thread_unref(&f->in_cdf);
 3723|  28.3k|    if (f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3723:9): [True: 4.67k, False: 23.7k]
  ------------------
 3724|  4.67k|        dav1d_cdf_thread_unref(&f->out_cdf);
 3725|   227k|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3725:21): [True: 198k, False: 28.3k]
  ------------------
 3726|   198k|        if (f->refp[i].p.frame_hdr)
  ------------------
  |  Branch (3726:13): [True: 0, False: 198k]
  ------------------
 3727|      0|            dav1d_thread_picture_unref(&f->refp[i]);
 3728|   198k|        dav1d_ref_dec(&f->ref_mvs_ref[i]);
 3729|   198k|    }
 3730|  28.3k|    if (c->n_fc == 1)
  ------------------
  |  Branch (3730:9): [True: 28.3k, False: 0]
  ------------------
 3731|  28.3k|        dav1d_thread_picture_unref(&c->out);
 3732|      0|    else
 3733|      0|        dav1d_thread_picture_unref(out_delayed);
 3734|  28.3k|    dav1d_picture_unref_internal(&f->cur);
 3735|  28.3k|    dav1d_thread_picture_unref(&f->sr_cur);
 3736|  28.3k|    dav1d_ref_dec(&f->mvs_ref);
 3737|  28.3k|    dav1d_ref_dec(&f->seq_hdr_ref);
 3738|  28.3k|    dav1d_ref_dec(&f->frame_hdr_ref);
 3739|  28.3k|    dav1d_data_props_copy(&c->cached_error_props, &c->in.m);
 3740|       |
 3741|  28.3k|    for (int i = 0; i < f->n_tile_data; i++)
  ------------------
  |  Branch (3741:21): [True: 0, False: 28.3k]
  ------------------
 3742|      0|        dav1d_data_unref_internal(&f->tile[i].data);
 3743|  28.3k|    f->n_tile_data = 0;
 3744|       |
 3745|  28.3k|    if (c->n_fc > 1)
  ------------------
  |  Branch (3745:9): [True: 0, False: 28.3k]
  ------------------
 3746|      0|        pthread_mutex_unlock(&c->task_thread.lock);
 3747|       |
 3748|  28.3k|    return res;
 3749|  45.8k|}
decode.c:reset_context:
 2392|   541k|static void reset_context(BlockContext *const ctx, const int keyframe, const int pass) {
 2393|   541k|    memset(ctx->intra, keyframe, sizeof(ctx->intra));
 2394|   541k|    memset(ctx->uvmode, DC_PRED, sizeof(ctx->uvmode));
 2395|   541k|    if (keyframe)
  ------------------
  |  Branch (2395:9): [True: 445k, False: 96.1k]
  ------------------
 2396|   445k|        memset(ctx->mode, DC_PRED, sizeof(ctx->mode));
 2397|       |
 2398|   541k|    if (pass == 2) return;
  ------------------
  |  Branch (2398:9): [True: 0, False: 541k]
  ------------------
 2399|       |
 2400|   541k|    memset(ctx->partition, 0, sizeof(ctx->partition));
 2401|   541k|    memset(ctx->skip, 0, sizeof(ctx->skip));
 2402|   541k|    memset(ctx->skip_mode, 0, sizeof(ctx->skip_mode));
 2403|   541k|    memset(ctx->tx_lpf_y, 2, sizeof(ctx->tx_lpf_y));
 2404|   541k|    memset(ctx->tx_lpf_uv, 1, sizeof(ctx->tx_lpf_uv));
 2405|   541k|    memset(ctx->tx_intra, -1, sizeof(ctx->tx_intra));
 2406|   541k|    memset(ctx->tx, TX_64X64, sizeof(ctx->tx));
 2407|   541k|    if (!keyframe) {
  ------------------
  |  Branch (2407:9): [True: 96.1k, False: 445k]
  ------------------
 2408|  96.1k|        memset(ctx->ref, -1, sizeof(ctx->ref));
 2409|  96.1k|        memset(ctx->comp_type, 0, sizeof(ctx->comp_type));
 2410|  96.1k|        memset(ctx->mode, NEARESTMV, sizeof(ctx->mode));
 2411|  96.1k|    }
 2412|   541k|    memset(ctx->lcoef, 0x40, sizeof(ctx->lcoef));
 2413|   541k|    memset(ctx->ccoef, 0x40, sizeof(ctx->ccoef));
 2414|   541k|    memset(ctx->filter, DAV1D_N_SWITCHABLE_FILTERS, sizeof(ctx->filter));
 2415|   541k|    memset(ctx->seg_pred, 0, sizeof(ctx->seg_pred));
 2416|   541k|    memset(ctx->pal_sz, 0, sizeof(ctx->pal_sz));
 2417|   541k|}
decode.c:decode_sb:
 2121|  2.94M|{
 2122|  2.94M|    const Dav1dFrameContext *const f = t->f;
 2123|  2.94M|    Dav1dTileState *const ts = t->ts;
 2124|  2.94M|    const int hsz = 16 >> bl;
 2125|  2.94M|    const int have_h_split = f->bw > t->bx + hsz;
 2126|  2.94M|    const int have_v_split = f->bh > t->by + hsz;
 2127|       |
 2128|  2.94M|    if (!have_h_split && !have_v_split) {
  ------------------
  |  Branch (2128:9): [True: 113k, False: 2.83M]
  |  Branch (2128:26): [True: 50.7k, False: 62.6k]
  ------------------
 2129|  50.7k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2129:9): [True: 50.7k, False: 0]
  ------------------
 2130|  50.7k|        return decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0));
  ------------------
  |  |   51|  50.7k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
 2131|  50.7k|    }
 2132|       |
 2133|  2.89M|    uint16_t *pc;
 2134|  2.89M|    enum BlockPartition bp;
 2135|  2.89M|    int ctx, bx8, by8;
 2136|  2.89M|    if (t->frame_thread.pass != 2) {
  ------------------
  |  Branch (2136:9): [True: 2.89M, False: 0]
  ------------------
 2137|  2.89M|        if (0 && bl == BL_64X64)
  ------------------
  |  Branch (2137:13): [Folded, False: 2.89M]
  |  Branch (2137:18): [True: 0, False: 0]
  ------------------
 2138|      0|            printf("poc=%d,y=%d,x=%d,bl=%d,r=%d\n",
 2139|      0|                   f->frame_hdr->frame_offset, t->by, t->bx, bl, ts->msac.rng);
 2140|  2.89M|        bx8 = (t->bx & 31) >> 1;
 2141|  2.89M|        by8 = (t->by & 31) >> 1;
 2142|  2.89M|        ctx = get_partition_ctx(t->a, &t->l, bl, by8, bx8);
 2143|  2.89M|        pc = ts->cdf.m.partition[bl][ctx];
 2144|  2.89M|    }
 2145|       |
 2146|  2.89M|    if (have_h_split && have_v_split) {
  ------------------
  |  Branch (2146:9): [True: 2.83M, False: 62.6k]
  |  Branch (2146:25): [True: 2.30M, False: 526k]
  ------------------
 2147|  2.30M|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2147:13): [True: 0, False: 2.30M]
  ------------------
 2148|      0|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2149|      0|            bp = b->bl == bl ? b->bp : PARTITION_SPLIT;
  ------------------
  |  Branch (2149:18): [True: 0, False: 0]
  ------------------
 2150|  2.30M|        } else {
 2151|  2.30M|            bp = dav1d_msac_decode_symbol_adapt16(&ts->msac, pc,
  ------------------
  |  |   57|  2.30M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 2152|  2.30M|                                                  dav1d_partition_type_count[bl]);
 2153|  2.30M|            if (f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I422 &&
  ------------------
  |  Branch (2153:17): [True: 20.3k, False: 2.28M]
  ------------------
 2154|  20.3k|                (bp == PARTITION_V || bp == PARTITION_V4 ||
  ------------------
  |  Branch (2154:18): [True: 330, False: 20.0k]
  |  Branch (2154:39): [True: 247, False: 19.8k]
  ------------------
 2155|  19.8k|                 bp == PARTITION_T_LEFT_SPLIT || bp == PARTITION_T_RIGHT_SPLIT))
  ------------------
  |  Branch (2155:18): [True: 205, False: 19.5k]
  |  Branch (2155:50): [True: 198, False: 19.4k]
  ------------------
 2156|    980|            {
 2157|    980|                return 1;
 2158|    980|            }
 2159|  2.30M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.30M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.30M]
  |  |  ------------------
  |  |   35|  2.30M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.30M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2160|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2161|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx, bp,
 2162|      0|                       ts->msac.rng);
 2163|  2.30M|        }
 2164|  2.30M|        const uint8_t *const b = dav1d_block_sizes[bl][bp];
 2165|       |
 2166|  2.30M|        switch (bp) {
 2167|   905k|        case PARTITION_NONE:
  ------------------
  |  Branch (2167:9): [True: 905k, False: 1.40M]
  ------------------
 2168|   905k|            if (decode_b(t, bl, b[0], PARTITION_NONE, node->o))
  ------------------
  |  Branch (2168:17): [True: 322, False: 904k]
  ------------------
 2169|    322|                return -1;
 2170|   904k|            break;
 2171|   904k|        case PARTITION_H:
  ------------------
  |  Branch (2171:9): [True: 267k, False: 2.04M]
  ------------------
 2172|   267k|            if (decode_b(t, bl, b[0], PARTITION_H, node->h[0]))
  ------------------
  |  Branch (2172:17): [True: 430, False: 266k]
  ------------------
 2173|    430|                return -1;
 2174|   266k|            t->by += hsz;
 2175|   266k|            if (decode_b(t, bl, b[0], PARTITION_H, node->h[1]))
  ------------------
  |  Branch (2175:17): [True: 207, False: 266k]
  ------------------
 2176|    207|                return -1;
 2177|   266k|            t->by -= hsz;
 2178|   266k|            break;
 2179|   174k|        case PARTITION_V:
  ------------------
  |  Branch (2179:9): [True: 174k, False: 2.13M]
  ------------------
 2180|   174k|            if (decode_b(t, bl, b[0], PARTITION_V, node->v[0]))
  ------------------
  |  Branch (2180:17): [True: 307, False: 173k]
  ------------------
 2181|    307|                return -1;
 2182|   173k|            t->bx += hsz;
 2183|   173k|            if (decode_b(t, bl, b[0], PARTITION_V, node->v[1]))
  ------------------
  |  Branch (2183:17): [True: 209, False: 173k]
  ------------------
 2184|    209|                return -1;
 2185|   173k|            t->bx -= hsz;
 2186|   173k|            break;
 2187|   545k|        case PARTITION_SPLIT:
  ------------------
  |  Branch (2187:9): [True: 545k, False: 1.76M]
  ------------------
 2188|   545k|            if (bl == BL_8X8) {
  ------------------
  |  Branch (2188:17): [True: 156k, False: 388k]
  ------------------
 2189|   156k|                const EdgeTip *const tip = (const EdgeTip *) node;
 2190|   156k|                assert(hsz == 1);
  ------------------
  |  Branch (2190:17): [True: 156k, False: 0]
  ------------------
 2191|   156k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2191:21): [True: 281, False: 156k]
  ------------------
 2192|    281|                    return -1;
 2193|   156k|                const enum Filter2d tl_filter = t->tl_4x4_filter;
 2194|   156k|                t->bx++;
 2195|   156k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[0]))
  ------------------
  |  Branch (2195:21): [True: 195, False: 156k]
  ------------------
 2196|    195|                    return -1;
 2197|   156k|                t->bx--;
 2198|   156k|                t->by++;
 2199|   156k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[1]))
  ------------------
  |  Branch (2199:21): [True: 222, False: 156k]
  ------------------
 2200|    222|                    return -1;
 2201|   156k|                t->bx++;
 2202|   156k|                t->tl_4x4_filter = tl_filter;
 2203|   156k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[2]))
  ------------------
  |  Branch (2203:21): [True: 197, False: 156k]
  ------------------
 2204|    197|                    return -1;
 2205|   156k|                t->bx--;
 2206|   156k|                t->by--;
 2207|   156k|#if ARCH_X86_64
 2208|   156k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (2208:21): [True: 0, False: 156k]
  ------------------
 2209|       |                    /* In 8-bit mode with 2-pass decoding the coefficient buffer
 2210|       |                     * can end up misaligned due to skips here. Work around
 2211|       |                     * the issue by explicitly realigning the buffer. */
 2212|      0|                    const int p = t->frame_thread.pass & 1;
 2213|      0|                    ts->frame_thread[p].cf =
 2214|      0|                        (void*)(((uintptr_t)ts->frame_thread[p].cf + 63) & ~63);
 2215|      0|                }
 2216|   156k|#endif
 2217|   388k|            } else {
 2218|   388k|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0)))
  ------------------
  |  |   51|   388k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2218:21): [True: 1.51k, False: 387k]
  ------------------
 2219|  1.51k|                    return 1;
 2220|   387k|                t->bx += hsz;
 2221|   387k|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 1)))
  ------------------
  |  |   51|   387k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2221:21): [True: 1.48k, False: 385k]
  ------------------
 2222|  1.48k|                    return 1;
 2223|   385k|                t->bx -= hsz;
 2224|   385k|                t->by += hsz;
 2225|   385k|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 2)))
  ------------------
  |  |   51|   385k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2225:21): [True: 566, False: 385k]
  ------------------
 2226|    566|                    return 1;
 2227|   385k|                t->bx += hsz;
 2228|   385k|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 3)))
  ------------------
  |  |   51|   385k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2228:21): [True: 1.04k, False: 384k]
  ------------------
 2229|  1.04k|                    return 1;
 2230|   384k|                t->bx -= hsz;
 2231|   384k|                t->by -= hsz;
 2232|   384k|            }
 2233|   540k|            break;
 2234|   540k|        case PARTITION_T_TOP_SPLIT: {
  ------------------
  |  Branch (2234:9): [True: 39.4k, False: 2.26M]
  ------------------
 2235|  39.4k|            if (decode_b(t, bl, b[0], PARTITION_T_TOP_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2235:17): [True: 370, False: 39.0k]
  ------------------
 2236|    370|                return -1;
 2237|  39.0k|            t->bx += hsz;
 2238|  39.0k|            if (decode_b(t, bl, b[0], PARTITION_T_TOP_SPLIT, node->v[1]))
  ------------------
  |  Branch (2238:17): [True: 196, False: 38.8k]
  ------------------
 2239|    196|                return -1;
 2240|  38.8k|            t->bx -= hsz;
 2241|  38.8k|            t->by += hsz;
 2242|  38.8k|            if (decode_b(t, bl, b[1], PARTITION_T_TOP_SPLIT, node->h[1]))
  ------------------
  |  Branch (2242:17): [True: 79, False: 38.7k]
  ------------------
 2243|     79|                return -1;
 2244|  38.7k|            t->by -= hsz;
 2245|  38.7k|            break;
 2246|  38.8k|        }
 2247|  43.1k|        case PARTITION_T_BOTTOM_SPLIT: {
  ------------------
  |  Branch (2247:9): [True: 43.1k, False: 2.26M]
  ------------------
 2248|  43.1k|            if (decode_b(t, bl, b[0], PARTITION_T_BOTTOM_SPLIT, node->h[0]))
  ------------------
  |  Branch (2248:17): [True: 265, False: 42.8k]
  ------------------
 2249|    265|                return -1;
 2250|  42.8k|            t->by += hsz;
 2251|  42.8k|            if (decode_b(t, bl, b[1], PARTITION_T_BOTTOM_SPLIT, node->v[0]))
  ------------------
  |  Branch (2251:17): [True: 195, False: 42.7k]
  ------------------
 2252|    195|                return -1;
 2253|  42.7k|            t->bx += hsz;
 2254|  42.7k|            if (decode_b(t, bl, b[1], PARTITION_T_BOTTOM_SPLIT, 0))
  ------------------
  |  Branch (2254:17): [True: 211, False: 42.4k]
  ------------------
 2255|    211|                return -1;
 2256|  42.4k|            t->bx -= hsz;
 2257|  42.4k|            t->by -= hsz;
 2258|  42.4k|            break;
 2259|  42.7k|        }
 2260|  29.6k|        case PARTITION_T_LEFT_SPLIT: {
  ------------------
  |  Branch (2260:9): [True: 29.6k, False: 2.27M]
  ------------------
 2261|  29.6k|            if (decode_b(t, bl, b[0], PARTITION_T_LEFT_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2261:17): [True: 249, False: 29.3k]
  ------------------
 2262|    249|                return -1;
 2263|  29.3k|            t->by += hsz;
 2264|  29.3k|            if (decode_b(t, bl, b[0], PARTITION_T_LEFT_SPLIT, node->h[1]))
  ------------------
  |  Branch (2264:17): [True: 197, False: 29.1k]
  ------------------
 2265|    197|                return -1;
 2266|  29.1k|            t->by -= hsz;
 2267|  29.1k|            t->bx += hsz;
 2268|  29.1k|            if (decode_b(t, bl, b[1], PARTITION_T_LEFT_SPLIT, node->v[1]))
  ------------------
  |  Branch (2268:17): [True: 195, False: 28.9k]
  ------------------
 2269|    195|                return -1;
 2270|  28.9k|            t->bx -= hsz;
 2271|  28.9k|            break;
 2272|  29.1k|        }
 2273|  50.8k|        case PARTITION_T_RIGHT_SPLIT: {
  ------------------
  |  Branch (2273:9): [True: 50.8k, False: 2.25M]
  ------------------
 2274|  50.8k|            if (decode_b(t, bl, b[0], PARTITION_T_RIGHT_SPLIT, node->v[0]))
  ------------------
  |  Branch (2274:17): [True: 199, False: 50.6k]
  ------------------
 2275|    199|                return -1;
 2276|  50.6k|            t->bx += hsz;
 2277|  50.6k|            if (decode_b(t, bl, b[1], PARTITION_T_RIGHT_SPLIT, node->h[0]))
  ------------------
  |  Branch (2277:17): [True: 196, False: 50.4k]
  ------------------
 2278|    196|                return -1;
 2279|  50.4k|            t->by += hsz;
 2280|  50.4k|            if (decode_b(t, bl, b[1], PARTITION_T_RIGHT_SPLIT, 0))
  ------------------
  |  Branch (2280:17): [True: 194, False: 50.2k]
  ------------------
 2281|    194|                return -1;
 2282|  50.2k|            t->by -= hsz;
 2283|  50.2k|            t->bx -= hsz;
 2284|  50.2k|            break;
 2285|  50.4k|        }
 2286|   104k|        case PARTITION_H4: {
  ------------------
  |  Branch (2286:9): [True: 104k, False: 2.20M]
  ------------------
 2287|   104k|            const EdgeBranch *const branch = (const EdgeBranch *) node;
 2288|   104k|            if (decode_b(t, bl, b[0], PARTITION_H4, node->h[0]))
  ------------------
  |  Branch (2288:17): [True: 199, False: 103k]
  ------------------
 2289|    199|                return -1;
 2290|   103k|            t->by += hsz >> 1;
 2291|   103k|            if (decode_b(t, bl, b[0], PARTITION_H4, branch->h4))
  ------------------
  |  Branch (2291:17): [True: 207, False: 103k]
  ------------------
 2292|    207|                return -1;
 2293|   103k|            t->by += hsz >> 1;
 2294|   103k|            if (decode_b(t, bl, b[0], PARTITION_H4, EDGE_ALL_LEFT_HAS_BOTTOM))
  ------------------
  |  Branch (2294:17): [True: 209, False: 103k]
  ------------------
 2295|    209|                return -1;
 2296|   103k|            t->by += hsz >> 1;
 2297|   103k|            if (t->by < f->bh)
  ------------------
  |  Branch (2297:17): [True: 99.2k, False: 4.15k]
  ------------------
 2298|  99.2k|                if (decode_b(t, bl, b[0], PARTITION_H4, node->h[1]))
  ------------------
  |  Branch (2298:21): [True: 257, False: 99.0k]
  ------------------
 2299|    257|                    return -1;
 2300|   103k|            t->by -= hsz * 3 >> 1;
 2301|   103k|            break;
 2302|   103k|        }
 2303|   148k|        case PARTITION_V4: {
  ------------------
  |  Branch (2303:9): [True: 148k, False: 2.15M]
  ------------------
 2304|   148k|            const EdgeBranch *const branch = (const EdgeBranch *) node;
 2305|   148k|            if (decode_b(t, bl, b[0], PARTITION_V4, node->v[0]))
  ------------------
  |  Branch (2305:17): [True: 365, False: 148k]
  ------------------
 2306|    365|                return -1;
 2307|   148k|            t->bx += hsz >> 1;
 2308|   148k|            if (decode_b(t, bl, b[0], PARTITION_V4, branch->v4))
  ------------------
  |  Branch (2308:17): [True: 624, False: 147k]
  ------------------
 2309|    624|                return -1;
 2310|   147k|            t->bx += hsz >> 1;
 2311|   147k|            if (decode_b(t, bl, b[0], PARTITION_V4, EDGE_ALL_TOP_HAS_RIGHT))
  ------------------
  |  Branch (2311:17): [True: 201, False: 147k]
  ------------------
 2312|    201|                return -1;
 2313|   147k|            t->bx += hsz >> 1;
 2314|   147k|            if (t->bx < f->bw)
  ------------------
  |  Branch (2314:17): [True: 143k, False: 3.93k]
  ------------------
 2315|   143k|                if (decode_b(t, bl, b[0], PARTITION_V4, node->v[1]))
  ------------------
  |  Branch (2315:21): [True: 199, False: 143k]
  ------------------
 2316|    199|                    return -1;
 2317|   147k|            t->bx -= hsz * 3 >> 1;
 2318|   147k|            break;
 2319|   147k|        }
 2320|      0|        default: assert(0);
  ------------------
  |  Branch (2320:9): [True: 0, False: 2.30M]
  |  Branch (2320:18): [Folded, False: 0]
  ------------------
 2321|  2.30M|        }
 2322|  2.30M|    } else if (have_h_split) {
  ------------------
  |  Branch (2322:16): [True: 526k, False: 62.6k]
  ------------------
 2323|   526k|        unsigned is_split;
 2324|   526k|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2324:13): [True: 0, False: 526k]
  ------------------
 2325|      0|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2326|      0|            is_split = b->bl != bl;
 2327|   526k|        } else {
 2328|   526k|            is_split = dav1d_msac_decode_bool(&ts->msac,
  ------------------
  |  |   54|   526k|#define dav1d_msac_decode_bool           dav1d_msac_decode_bool_sse2
  ------------------
 2329|   526k|                           gather_top_partition_prob(pc, bl));
 2330|   526k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   526k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 526k]
  |  |  ------------------
  |  |   35|   526k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   526k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2331|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2332|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx,
 2333|      0|                       is_split ? PARTITION_SPLIT : PARTITION_H, ts->msac.rng);
  ------------------
  |  Branch (2333:24): [True: 0, False: 0]
  ------------------
 2334|   526k|        }
 2335|       |
 2336|   526k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2336:9): [True: 526k, False: 0]
  ------------------
 2337|   526k|        if (is_split) {
  ------------------
  |  Branch (2337:13): [True: 341k, False: 184k]
  ------------------
 2338|   341k|            bp = PARTITION_SPLIT;
 2339|   341k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0))) return 1;
  ------------------
  |  |   51|   341k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2339:17): [True: 4.38k, False: 337k]
  ------------------
 2340|   337k|            t->bx += hsz;
 2341|   337k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 1))) return 1;
  ------------------
  |  |   51|   337k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2341:17): [True: 3.11k, False: 334k]
  ------------------
 2342|   334k|            t->bx -= hsz;
 2343|   334k|        } else {
 2344|   184k|            bp = PARTITION_H;
 2345|   184k|            if (decode_b(t, bl, dav1d_block_sizes[bl][PARTITION_H][0],
  ------------------
  |  Branch (2345:17): [True: 256, False: 183k]
  ------------------
 2346|   184k|                         PARTITION_H, node->h[0]))
 2347|    256|                return -1;
 2348|   184k|        }
 2349|   526k|    } else {
 2350|  62.6k|        assert(have_v_split);
  ------------------
  |  Branch (2350:9): [True: 62.6k, False: 0]
  ------------------
 2351|  62.6k|        unsigned is_split;
 2352|  62.6k|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2352:13): [True: 0, False: 62.6k]
  ------------------
 2353|      0|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2354|      0|            is_split = b->bl != bl;
 2355|  62.6k|        } else {
 2356|  62.6k|            is_split = dav1d_msac_decode_bool(&ts->msac,
  ------------------
  |  |   54|  62.6k|#define dav1d_msac_decode_bool           dav1d_msac_decode_bool_sse2
  ------------------
 2357|  62.6k|                           gather_left_partition_prob(pc, bl));
 2358|  62.6k|            if (f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I422 && !is_split)
  ------------------
  |  Branch (2358:17): [True: 2.23k, False: 60.4k]
  |  Branch (2358:63): [True: 332, False: 1.90k]
  ------------------
 2359|    332|                return 1;
 2360|  62.3k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  62.3k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 62.3k]
  |  |  ------------------
  |  |   35|  62.3k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  62.3k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2361|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2362|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx,
 2363|      0|                       is_split ? PARTITION_SPLIT : PARTITION_V, ts->msac.rng);
  ------------------
  |  Branch (2363:24): [True: 0, False: 0]
  ------------------
 2364|  62.3k|        }
 2365|       |
 2366|  62.6k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2366:9): [True: 62.3k, False: 0]
  ------------------
 2367|  62.3k|        if (is_split) {
  ------------------
  |  Branch (2367:13): [True: 35.8k, False: 26.4k]
  ------------------
 2368|  35.8k|            bp = PARTITION_SPLIT;
 2369|  35.8k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0))) return 1;
  ------------------
  |  |   51|  35.8k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2369:17): [True: 3.00k, False: 32.8k]
  ------------------
 2370|  32.8k|            t->by += hsz;
 2371|  32.8k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 2))) return 1;
  ------------------
  |  |   51|  32.8k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2371:17): [True: 1.65k, False: 31.2k]
  ------------------
 2372|  31.2k|            t->by -= hsz;
 2373|  31.2k|        } else {
 2374|  26.4k|            bp = PARTITION_V;
 2375|  26.4k|            if (decode_b(t, bl, dav1d_block_sizes[bl][PARTITION_V][0],
  ------------------
  |  Branch (2375:17): [True: 344, False: 26.0k]
  ------------------
 2376|  26.4k|                         PARTITION_V, node->v[0]))
 2377|    344|                return -1;
 2378|  26.4k|        }
 2379|  62.3k|    }
 2380|       |
 2381|  2.87M|    if (t->frame_thread.pass != 2 && (bp != PARTITION_SPLIT || bl == BL_8X8)) {
  ------------------
  |  Branch (2381:9): [True: 2.87M, False: 0]
  |  Branch (2381:39): [True: 1.96M, False: 906k]
  |  Branch (2381:64): [True: 156k, False: 750k]
  ------------------
 2382|  2.12M|#define set_ctx(rep_macro) \
 2383|  2.12M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
 2384|  2.12M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
 2385|  2.12M|        case_set_upto16(ulog2(hsz));
  ------------------
  |  |   80|  2.12M|    switch (var) { \
  |  |   81|   596k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 2383|   596k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|   596k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   596k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   596k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|   596k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   596k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 596k, False: 1.52M]
  |  |  ------------------
  |  |   82|   680k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 2383|   680k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|   680k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   680k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   680k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|   680k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   680k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 680k, False: 1.44M]
  |  |  ------------------
  |  |   83|   403k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 2383|   403k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|   403k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   403k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   403k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|   403k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   403k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 403k, False: 1.71M]
  |  |  ------------------
  |  |   84|   283k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 2383|   283k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|   283k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   283k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   283k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|   283k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   283k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 283k, False: 1.83M]
  |  |  ------------------
  |  |   85|   158k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 2383|   158k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|   158k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   158k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   158k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   158k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 158k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   158k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|   158k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   158k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   158k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   158k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 158k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 158k, False: 1.96M]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 2.12M]
  |  |  ------------------
  |  |   87|  2.12M|    }
  ------------------
  |  Branch (2385:9): [Folded, False: 0]
  ------------------
 2386|  2.12M|#undef set_ctx
 2387|  2.12M|    }
 2388|       |
 2389|  2.87M|    return 0;
 2390|  2.87M|}
decode.c:decode_b:
  687|  4.10M|                    const enum EdgeFlags intra_edge_flags) {
  688|  4.10M|    Dav1dTileState *const ts = t->ts;
  689|  4.10M|    const Dav1dFrameContext *const f = t->f;
  690|  4.10M|    Av1Block b_mem, *const b = t->frame_thread.pass ?
  ------------------
  |  Branch (690:32): [True: 0, False: 4.10M]
  ------------------
  691|  4.10M|        &f->frame_thread.b[t->by * f->b4_stride + t->bx] : &b_mem;
  692|  4.10M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  693|  4.10M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  694|  4.10M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  695|  4.10M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  696|  4.10M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
  697|  4.10M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  698|  4.10M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
  699|  4.10M|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
  700|  4.10M|    const int have_left = t->bx > ts->tiling.col_start;
  701|  4.10M|    const int have_top = t->by > ts->tiling.row_start;
  702|  4.10M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (702:28): [True: 2.55M, False: 1.55M]
  ------------------
  703|  2.55M|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (703:29): [True: 2.39M, False: 156k]
  |  Branch (703:45): [True: 77.9k, False: 78.1k]
  ------------------
  704|  2.47M|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (704:29): [True: 2.37M, False: 101k]
  |  Branch (704:45): [True: 50.7k, False: 50.7k]
  ------------------
  705|       |
  706|  4.10M|    if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (706:9): [True: 0, False: 4.10M]
  ------------------
  707|      0|        if (b->intra) {
  ------------------
  |  Branch (707:13): [True: 0, False: 0]
  ------------------
  708|      0|            f->bd_fn.recon_b_intra(t, bs, intra_edge_flags, b);
  709|       |
  710|      0|            const enum IntraPredMode y_mode_nofilt =
  711|      0|                b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
  ------------------
  |  Branch (711:17): [True: 0, False: 0]
  ------------------
  712|      0|#define set_ctx(rep_macro) \
  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  714|      0|            rep_macro(edge->intra, off, 1)
  715|      0|            BlockContext *edge = t->a;
  716|      0|            for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (716:40): [True: 0, False: 0]
  ------------------
  717|      0|                case_set(b_dim[2 + i]);
  ------------------
  |  |   70|      0|    switch (var) { \
  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|      0|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|      0|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|      0|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|      0|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|      0|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|      0|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|      0|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|      0|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|      0|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|      0|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|      0|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|      0|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  |  713|      0|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|      0|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|      0|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|      0|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|      0|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|      0|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   78|      0|    }
  ------------------
  |  Branch (717:17): [Folded, False: 0]
  ------------------
  718|      0|            }
  719|      0|#undef set_ctx
  720|      0|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  721|      0|                refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
  722|      0|                for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (722:33): [True: 0, False: 0]
  ------------------
  723|      0|                    r[x].ref.ref[0] = 0;
  724|      0|                    r[x].bs = bs;
  725|      0|                }
  726|      0|                refmvs_block *const *rr = &t->rt.r[(t->by & 31) + 5];
  727|      0|                for (int y = 0; y < bh4 - 1; y++) {
  ------------------
  |  Branch (727:33): [True: 0, False: 0]
  ------------------
  728|      0|                    rr[y][t->bx + bw4 - 1].ref.ref[0] = 0;
  729|      0|                    rr[y][t->bx + bw4 - 1].bs = bs;
  730|      0|                }
  731|      0|            }
  732|       |
  733|      0|            if (has_chroma) {
  ------------------
  |  Branch (733:17): [True: 0, False: 0]
  ------------------
  734|      0|                uint8_t uv_mode = b->uv_mode;
  735|      0|                dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
  736|      0|                dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
  737|      0|            }
  738|      0|        } else {
  739|      0|            if (IS_INTER_OR_SWITCH(f->frame_hdr) /* not intrabc */ &&
  ------------------
  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  740|      0|                b->comp_type == COMP_INTER_NONE && b->motion_mode == MM_WARP)
  ------------------
  |  Branch (740:17): [True: 0, False: 0]
  |  Branch (740:52): [True: 0, False: 0]
  ------------------
  741|      0|            {
  742|      0|                if (b->matrix[0] == INT16_MIN) {
  ------------------
  |  Branch (742:21): [True: 0, False: 0]
  ------------------
  743|      0|                    t->warpmv.type = DAV1D_WM_TYPE_IDENTITY;
  744|      0|                } else {
  745|      0|                    t->warpmv.type = DAV1D_WM_TYPE_AFFINE;
  746|      0|                    t->warpmv.matrix[2] = b->matrix[0] + 0x10000;
  747|      0|                    t->warpmv.matrix[3] = b->matrix[1];
  748|      0|                    t->warpmv.matrix[4] = b->matrix[2];
  749|      0|                    t->warpmv.matrix[5] = b->matrix[3] + 0x10000;
  750|      0|                    dav1d_set_affine_mv2d(bw4, bh4, b->mv2d, &t->warpmv,
  751|      0|                                          t->bx, t->by);
  752|      0|                    dav1d_get_shear_params(&t->warpmv);
  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  754|      0|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|      0|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 0]
  |  |  ------------------
  |  |   35|      0|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|      0|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  755|      0|                        printf("[ %c%x %c%x %c%x\n  %c%x %c%x %c%x ]\n"
  756|      0|                               "alpha=%c%x, beta=%c%x, gamma=%c%x, delta=%c%x, mv=y:%d,x:%d\n",
  757|      0|                               signabs(t->warpmv.matrix[0]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  758|      0|                               signabs(t->warpmv.matrix[1]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  759|      0|                               signabs(t->warpmv.matrix[2]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  760|      0|                               signabs(t->warpmv.matrix[3]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  761|      0|                               signabs(t->warpmv.matrix[4]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  762|      0|                               signabs(t->warpmv.matrix[5]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  763|      0|                               signabs(t->warpmv.u.p.alpha),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  764|      0|                               signabs(t->warpmv.u.p.beta),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  765|      0|                               signabs(t->warpmv.u.p.gamma),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  766|      0|                               signabs(t->warpmv.u.p.delta),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  767|      0|                               b->mv2d.y, b->mv2d.x);
  768|      0|#undef signabs
  769|      0|                }
  770|      0|            }
  771|      0|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (771:17): [True: 0, False: 0]
  ------------------
  772|       |
  773|      0|            const uint8_t *const filter = dav1d_filter_dir[b->filter2d];
  774|      0|            BlockContext *edge = t->a;
  775|      0|            for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (775:40): [True: 0, False: 0]
  ------------------
  776|      0|#define set_ctx(rep_macro) \
  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  779|      0|                rep_macro(edge->intra, off, 0)
  780|      0|                case_set(b_dim[2 + i]);
  ------------------
  |  |   70|      0|    switch (var) { \
  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|      0|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|      0|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|      0|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|      0|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|      0|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|      0|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|      0|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|      0|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|      0|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|      0|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|      0|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|      0|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|      0|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|      0|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|      0|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|      0|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|      0|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|      0|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|      0|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|      0|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|      0|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|      0|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|      0|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  |  777|      0|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|      0|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|      0|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|      0|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|      0|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|      0|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|      0|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|      0|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|      0|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|      0|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|      0|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 0]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 0]
  |  |  ------------------
  |  |   78|      0|    }
  ------------------
  |  Branch (780:17): [Folded, False: 0]
  ------------------
  781|      0|#undef set_ctx
  782|      0|            }
  783|       |
  784|      0|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  785|      0|                refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
  786|      0|                const int ref1 = b->ref[0] + 1;
  787|      0|                const union mv mv1 = b->mv[0];
  788|      0|                for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (788:33): [True: 0, False: 0]
  ------------------
  789|      0|                    r[x].ref.ref[0] = ref1;
  790|      0|                    r[x].mv.mv[0] = mv1;
  791|      0|                    r[x].bs = bs;
  792|      0|                }
  793|      0|                refmvs_block *const *rr = &t->rt.r[(t->by & 31) + 5];
  794|      0|                for (int y = 0; y < bh4 - 1; y++) {
  ------------------
  |  Branch (794:33): [True: 0, False: 0]
  ------------------
  795|      0|                    rr[y][t->bx + bw4 - 1].ref.ref[0] = ref1;
  796|      0|                    rr[y][t->bx + bw4 - 1].mv.mv[0] = mv1;
  797|      0|                    rr[y][t->bx + bw4 - 1].bs = bs;
  798|      0|                }
  799|      0|            }
  800|       |
  801|      0|            if (has_chroma) {
  ------------------
  |  Branch (801:17): [True: 0, False: 0]
  ------------------
  802|      0|                dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
  803|      0|                dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
  804|      0|            }
  805|      0|        }
  806|      0|        return 0;
  807|      0|    }
  808|       |
  809|  4.10M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
  810|       |
  811|  4.10M|    b->bl = bl;
  812|  4.10M|    b->bp = bp;
  813|  4.10M|    b->bs = bs;
  814|       |
  815|  4.10M|    const Dav1dSegmentationData *seg = NULL;
  816|       |
  817|       |    // segment_id (if seg_feature for skip/ref/gmv is enabled)
  818|  4.10M|    int seg_pred = 0;
  819|  4.10M|    if (f->frame_hdr->segmentation.enabled) {
  ------------------
  |  Branch (819:9): [True: 1.39M, False: 2.71M]
  ------------------
  820|  1.39M|        if (!f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (820:13): [True: 313k, False: 1.07M]
  ------------------
  821|   313k|            if (f->prev_segmap) {
  ------------------
  |  Branch (821:17): [True: 206k, False: 106k]
  ------------------
  822|   206k|                unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx, w4, h4,
  823|   206k|                                                       f->prev_segmap,
  824|   206k|                                                       f->b4_stride);
  825|   206k|                if (seg_id >= 8) return -1;
  ------------------
  |  Branch (825:21): [True: 0, False: 206k]
  ------------------
  826|   206k|                b->seg_id = seg_id;
  827|   206k|            } else {
  828|   106k|                b->seg_id = 0;
  829|   106k|            }
  830|   313k|            seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  831|  1.07M|        } else if (f->frame_hdr->segmentation.seg_data.preskip) {
  ------------------
  |  Branch (831:20): [True: 514k, False: 565k]
  ------------------
  832|   514k|            if (f->frame_hdr->segmentation.temporal &&
  ------------------
  |  Branch (832:17): [True: 42.0k, False: 472k]
  ------------------
  833|  42.0k|                (seg_pred = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  42.0k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (833:17): [True: 25.7k, False: 16.2k]
  ------------------
  834|  42.0k|                                ts->cdf.m.seg_pred[t->a->seg_pred[bx4] +
  835|  42.0k|                                t->l.seg_pred[by4]])))
  836|  25.7k|            {
  837|       |                // temporal predicted seg_id
  838|  25.7k|                if (f->prev_segmap) {
  ------------------
  |  Branch (838:21): [True: 11.4k, False: 14.3k]
  ------------------
  839|  11.4k|                    unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx,
  840|  11.4k|                                                           w4, h4,
  841|  11.4k|                                                           f->prev_segmap,
  842|  11.4k|                                                           f->b4_stride);
  843|  11.4k|                    if (seg_id >= 8) return -1;
  ------------------
  |  Branch (843:25): [True: 0, False: 11.4k]
  ------------------
  844|  11.4k|                    b->seg_id = seg_id;
  845|  14.3k|                } else {
  846|  14.3k|                    b->seg_id = 0;
  847|  14.3k|                }
  848|   488k|            } else {
  849|   488k|                int seg_ctx;
  850|   488k|                const unsigned pred_seg_id =
  851|   488k|                    get_cur_frame_segid(t->by, t->bx, have_top, have_left,
  852|   488k|                                        &seg_ctx, f->cur_segmap, f->b4_stride);
  853|   488k|                const unsigned diff = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   488k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  854|   488k|                                          ts->cdf.m.seg_id[seg_ctx],
  855|   488k|                                          DAV1D_MAX_SEGMENTS - 1);
  ------------------
  |  |   43|   488k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  856|   488k|                const unsigned last_active_seg_id =
  857|   488k|                    f->frame_hdr->segmentation.seg_data.last_active_segid;
  858|   488k|                b->seg_id = neg_deinterleave(diff, pred_seg_id,
  859|   488k|                                             last_active_seg_id + 1);
  860|   488k|                if (b->seg_id > last_active_seg_id) b->seg_id = 0; // error?
  ------------------
  |  Branch (860:21): [True: 31.4k, False: 457k]
  ------------------
  861|   488k|                if (b->seg_id >= DAV1D_MAX_SEGMENTS) b->seg_id = 0; // error?
  ------------------
  |  |   43|   488k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (861:21): [True: 0, False: 488k]
  ------------------
  862|   488k|            }
  863|       |
  864|   514k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   514k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 514k]
  |  |  ------------------
  |  |   35|   514k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   514k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  865|      0|                printf("Post-segid[preskip;%d]: r=%d\n",
  866|      0|                       b->seg_id, ts->msac.rng);
  867|       |
  868|   514k|            seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  869|   514k|        }
  870|  2.71M|    } else {
  871|  2.71M|        b->seg_id = 0;
  872|  2.71M|    }
  873|       |
  874|       |    // skip_mode
  875|  4.10M|    if ((!seg || (!seg->globalmv && seg->ref == -1 && !seg->skip)) &&
  ------------------
  |  Branch (875:10): [True: 3.28M, False: 827k]
  |  Branch (875:19): [True: 501k, False: 326k]
  |  Branch (875:37): [True: 209k, False: 292k]
  |  Branch (875:55): [True: 143k, False: 66.2k]
  ------------------
  876|  3.42M|        f->frame_hdr->skip_mode_enabled && imin(bw4, bh4) > 1)
  ------------------
  |  Branch (876:9): [True: 57.9k, False: 3.36M]
  |  Branch (876:44): [True: 45.6k, False: 12.2k]
  ------------------
  877|  45.6k|    {
  878|  45.6k|        const int smctx = t->a->skip_mode[bx4] + t->l.skip_mode[by4];
  879|  45.6k|        b->skip_mode = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  45.6k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  880|  45.6k|                           ts->cdf.m.skip_mode[smctx]);
  881|  45.6k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  45.6k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 45.6k]
  |  |  ------------------
  |  |   35|  45.6k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  45.6k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  882|      0|            printf("Post-skipmode[%d]: r=%d\n", b->skip_mode, ts->msac.rng);
  883|  4.06M|    } else {
  884|  4.06M|        b->skip_mode = 0;
  885|  4.06M|    }
  886|       |
  887|       |    // skip
  888|  4.10M|    if (b->skip_mode || (seg && seg->skip)) {
  ------------------
  |  Branch (888:9): [True: 14.8k, False: 4.09M]
  |  Branch (888:26): [True: 827k, False: 3.26M]
  |  Branch (888:33): [True: 593k, False: 233k]
  ------------------
  889|   608k|        b->skip = 1;
  890|  3.50M|    } else {
  891|  3.50M|        const int sctx = t->a->skip[bx4] + t->l.skip[by4];
  892|  3.50M|        b->skip = dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.skip[sctx]);
  ------------------
  |  |   52|  3.50M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  893|  3.50M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  3.50M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 3.50M]
  |  |  ------------------
  |  |   35|  3.50M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  3.50M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  894|      0|            printf("Post-skip[%d]: r=%d\n", b->skip, ts->msac.rng);
  895|  3.50M|    }
  896|       |
  897|       |    // segment_id
  898|  4.10M|    if (f->frame_hdr->segmentation.enabled &&
  ------------------
  |  Branch (898:9): [True: 1.39M, False: 2.71M]
  ------------------
  899|  1.39M|        f->frame_hdr->segmentation.update_map &&
  ------------------
  |  Branch (899:9): [True: 1.07M, False: 313k]
  ------------------
  900|  1.07M|        !f->frame_hdr->segmentation.seg_data.preskip)
  ------------------
  |  Branch (900:9): [True: 565k, False: 514k]
  ------------------
  901|   565k|    {
  902|   565k|        if (!b->skip && f->frame_hdr->segmentation.temporal &&
  ------------------
  |  Branch (902:13): [True: 216k, False: 348k]
  |  Branch (902:25): [True: 9.76k, False: 207k]
  ------------------
  903|  9.76k|            (seg_pred = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  9.76k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (903:13): [True: 4.98k, False: 4.78k]
  ------------------
  904|  9.76k|                            ts->cdf.m.seg_pred[t->a->seg_pred[bx4] +
  905|  9.76k|                            t->l.seg_pred[by4]])))
  906|  4.98k|        {
  907|       |            // temporal predicted seg_id
  908|  4.98k|            if (f->prev_segmap) {
  ------------------
  |  Branch (908:17): [True: 1.22k, False: 3.76k]
  ------------------
  909|  1.22k|                unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx, w4, h4,
  910|  1.22k|                                                       f->prev_segmap,
  911|  1.22k|                                                       f->b4_stride);
  912|  1.22k|                if (seg_id >= 8) return -1;
  ------------------
  |  Branch (912:21): [True: 0, False: 1.22k]
  ------------------
  913|  1.22k|                b->seg_id = seg_id;
  914|  3.76k|            } else {
  915|  3.76k|                b->seg_id = 0;
  916|  3.76k|            }
  917|   560k|        } else {
  918|   560k|            int seg_ctx;
  919|   560k|            const unsigned pred_seg_id =
  920|   560k|                get_cur_frame_segid(t->by, t->bx, have_top, have_left,
  921|   560k|                                    &seg_ctx, f->cur_segmap, f->b4_stride);
  922|   560k|            if (b->skip) {
  ------------------
  |  Branch (922:17): [True: 348k, False: 211k]
  ------------------
  923|   348k|                b->seg_id = pred_seg_id;
  924|   348k|            } else {
  925|   211k|                const unsigned diff = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   211k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  926|   211k|                                          ts->cdf.m.seg_id[seg_ctx],
  927|   211k|                                          DAV1D_MAX_SEGMENTS - 1);
  ------------------
  |  |   43|   211k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  928|   211k|                const unsigned last_active_seg_id =
  929|   211k|                    f->frame_hdr->segmentation.seg_data.last_active_segid;
  930|   211k|                b->seg_id = neg_deinterleave(diff, pred_seg_id,
  931|   211k|                                             last_active_seg_id + 1);
  932|   211k|                if (b->seg_id > last_active_seg_id) b->seg_id = 0; // error?
  ------------------
  |  Branch (932:21): [True: 5.49k, False: 206k]
  ------------------
  933|   211k|            }
  934|   560k|            if (b->seg_id >= DAV1D_MAX_SEGMENTS) b->seg_id = 0; // error?
  ------------------
  |  |   43|   560k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (934:17): [True: 9.09k, False: 551k]
  ------------------
  935|   560k|        }
  936|       |
  937|   565k|        seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  938|       |
  939|   565k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   565k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 565k]
  |  |  ------------------
  |  |   35|   565k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   565k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  940|      0|            printf("Post-segid[postskip;%d]: r=%d\n",
  941|      0|                   b->seg_id, ts->msac.rng);
  942|   565k|    }
  943|       |
  944|       |    // cdef index
  945|  4.10M|    if (!b->skip) {
  ------------------
  |  Branch (945:9): [True: 2.21M, False: 1.89M]
  ------------------
  946|  2.21M|        const int idx = f->seq_hdr->sb128 ? ((t->bx & 16) >> 4) +
  ------------------
  |  Branch (946:25): [True: 927k, False: 1.28M]
  ------------------
  947|  1.28M|                                           ((t->by & 16) >> 3) : 0;
  948|  2.21M|        if (t->cur_sb_cdef_idx_ptr[idx] == -1) {
  ------------------
  |  Branch (948:13): [True: 362k, False: 1.85M]
  ------------------
  949|   362k|            const int v = dav1d_msac_decode_bools(&ts->msac,
  950|   362k|                              f->frame_hdr->cdef.n_bits);
  951|   362k|            t->cur_sb_cdef_idx_ptr[idx] = v;
  952|   362k|            if (bw4 > 16) t->cur_sb_cdef_idx_ptr[idx + 1] = v;
  ------------------
  |  Branch (952:17): [True: 30.4k, False: 331k]
  ------------------
  953|   362k|            if (bh4 > 16) t->cur_sb_cdef_idx_ptr[idx + 2] = v;
  ------------------
  |  Branch (953:17): [True: 22.2k, False: 340k]
  ------------------
  954|   362k|            if (bw4 == 32 && bh4 == 32) t->cur_sb_cdef_idx_ptr[idx + 3] = v;
  ------------------
  |  Branch (954:17): [True: 30.4k, False: 331k]
  |  Branch (954:30): [True: 18.9k, False: 11.4k]
  ------------------
  955|       |
  956|   362k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   362k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 362k]
  |  |  ------------------
  |  |   35|   362k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   362k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  957|      0|                printf("Post-cdef_idx[%d]: r=%d\n",
  958|      0|                        *t->cur_sb_cdef_idx_ptr, ts->msac.rng);
  959|   362k|        }
  960|  2.21M|    }
  961|       |
  962|       |    // delta-q/lf
  963|  4.10M|    if (!((t->bx | t->by) & (31 >> !f->seq_hdr->sb128))) {
  ------------------
  |  Branch (963:9): [True: 600k, False: 3.50M]
  ------------------
  964|   600k|        const int prev_qidx = ts->last_qidx;
  965|   600k|        const int have_delta_q = f->frame_hdr->delta.q.present &&
  ------------------
  |  Branch (965:34): [True: 187k, False: 413k]
  ------------------
  966|   187k|            (bs != (f->seq_hdr->sb128 ? BS_128x128 : BS_64x64) || !b->skip);
  ------------------
  |  Branch (966:14): [True: 115k, False: 72.4k]
  |  Branch (966:21): [True: 34.3k, False: 153k]
  |  Branch (966:67): [True: 11.1k, False: 61.2k]
  ------------------
  967|       |
  968|   600k|        uint32_t prev_delta_lf = ts->last_delta_lf.u32;
  969|       |
  970|   600k|        if (have_delta_q) {
  ------------------
  |  Branch (970:13): [True: 126k, False: 474k]
  ------------------
  971|   126k|            int delta_q = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   126k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  972|   126k|                                                          ts->cdf.m.delta_q, 3);
  973|   126k|            if (delta_q == 3) {
  ------------------
  |  Branch (973:17): [True: 24.4k, False: 101k]
  ------------------
  974|  24.4k|                const int n_bits = 1 + dav1d_msac_decode_bools(&ts->msac, 3);
  975|  24.4k|                delta_q = dav1d_msac_decode_bools(&ts->msac, n_bits) +
  976|  24.4k|                          1 + (1 << n_bits);
  977|  24.4k|            }
  978|   126k|            if (delta_q) {
  ------------------
  |  Branch (978:17): [True: 36.6k, False: 89.6k]
  ------------------
  979|  36.6k|                if (dav1d_msac_decode_bool_equi(&ts->msac)) delta_q = -delta_q;
  ------------------
  |  |   53|  36.6k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (979:21): [True: 29.5k, False: 7.02k]
  ------------------
  980|  36.6k|                delta_q *= 1 << f->frame_hdr->delta.q.res_log2;
  981|  36.6k|            }
  982|   126k|            ts->last_qidx = iclip(ts->last_qidx + delta_q, 1, 255);
  983|   126k|            if (have_delta_q && DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   126k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 126k]
  |  |  ------------------
  |  |   35|   126k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   126k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (983:17): [True: 126k, False: 0]
  ------------------
  984|      0|                printf("Post-delta_q[%d->%d]: r=%d\n",
  985|      0|                       delta_q, ts->last_qidx, ts->msac.rng);
  986|       |
  987|   126k|            if (f->frame_hdr->delta.lf.present) {
  ------------------
  |  Branch (987:17): [True: 45.0k, False: 81.2k]
  ------------------
  988|  45.0k|                const int n_lfs = f->frame_hdr->delta.lf.multi ?
  ------------------
  |  Branch (988:35): [True: 38.3k, False: 6.66k]
  ------------------
  989|  38.3k|                    f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 ? 4 : 2 : 1;
  ------------------
  |  Branch (989:21): [True: 29.2k, False: 9.14k]
  ------------------
  990|       |
  991|   186k|                for (int i = 0; i < n_lfs; i++) {
  ------------------
  |  Branch (991:33): [True: 141k, False: 45.0k]
  ------------------
  992|   141k|                    int delta_lf = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   141k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  993|   141k|                        ts->cdf.m.delta_lf[i + f->frame_hdr->delta.lf.multi], 3);
  994|   141k|                    if (delta_lf == 3) {
  ------------------
  |  Branch (994:25): [True: 10.5k, False: 131k]
  ------------------
  995|  10.5k|                        const int n_bits = 1 + dav1d_msac_decode_bools(&ts->msac, 3);
  996|  10.5k|                        delta_lf = dav1d_msac_decode_bools(&ts->msac, n_bits) +
  997|  10.5k|                                   1 + (1 << n_bits);
  998|  10.5k|                    }
  999|   141k|                    if (delta_lf) {
  ------------------
  |  Branch (999:25): [True: 29.3k, False: 112k]
  ------------------
 1000|  29.3k|                        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|  29.3k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (1000:29): [True: 20.1k, False: 9.12k]
  ------------------
 1001|  20.1k|                            delta_lf = -delta_lf;
 1002|  29.3k|                        delta_lf *= 1 << f->frame_hdr->delta.lf.res_log2;
 1003|  29.3k|                    }
 1004|   141k|                    ts->last_delta_lf.i8[i] =
 1005|   141k|                        iclip(ts->last_delta_lf.i8[i] + delta_lf, -63, 63);
 1006|   141k|                    if (have_delta_q && DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   141k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 141k]
  |  |  ------------------
  |  |   35|   141k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   141k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (1006:25): [True: 141k, False: 0]
  ------------------
 1007|      0|                        printf("Post-delta_lf[%d:%d]: r=%d\n", i, delta_lf,
 1008|      0|                               ts->msac.rng);
 1009|   141k|                }
 1010|  45.0k|            }
 1011|   126k|        }
 1012|   600k|        if (ts->last_qidx == f->frame_hdr->quant.yac) {
  ------------------
  |  Branch (1012:13): [True: 505k, False: 95.7k]
  ------------------
 1013|       |            // assign frame-wide q values to this sb
 1014|   505k|            ts->dq = f->dq;
 1015|   505k|        } else if (ts->last_qidx != prev_qidx) {
  ------------------
  |  Branch (1015:20): [True: 11.0k, False: 84.7k]
  ------------------
 1016|       |            // find sb-specific quant parameters
 1017|  11.0k|            init_quant_tables(f->seq_hdr, f->frame_hdr, ts->last_qidx, ts->dqmem);
 1018|  11.0k|            ts->dq = ts->dqmem;
 1019|  11.0k|        }
 1020|   600k|        if (!ts->last_delta_lf.u32) {
  ------------------
  |  Branch (1020:13): [True: 551k, False: 48.9k]
  ------------------
 1021|       |            // assign frame-wide lf values to this sb
 1022|   551k|            ts->lflvl = f->lf.lvl;
 1023|   551k|        } else if (ts->last_delta_lf.u32 != prev_delta_lf) {
  ------------------
  |  Branch (1023:20): [True: 15.0k, False: 33.9k]
  ------------------
 1024|       |            // find sb-specific lf lvl parameters
 1025|  15.0k|            ts->lflvl = ts->lflvlmem;
 1026|  15.0k|            dav1d_calc_lf_values(ts->lflvlmem, f->frame_hdr, ts->last_delta_lf.i8);
 1027|  15.0k|        }
 1028|   600k|    }
 1029|       |
 1030|  4.10M|    if (b->skip_mode) {
  ------------------
  |  Branch (1030:9): [True: 14.8k, False: 4.09M]
  ------------------
 1031|  14.8k|        b->intra = 0;
 1032|  4.09M|    } else if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  4.09M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 1.16M, False: 2.93M]
  |  |  ------------------
  ------------------
 1033|  1.16M|        if (seg && (seg->ref >= 0 || seg->globalmv)) {
  ------------------
  |  Branch (1033:13): [True: 498k, False: 664k]
  |  Branch (1033:21): [True: 331k, False: 167k]
  |  Branch (1033:38): [True: 56.4k, False: 111k]
  ------------------
 1034|   387k|            b->intra = !seg->ref;
 1035|   775k|        } else {
 1036|   775k|            const int ictx = get_intra_ctx(t->a, &t->l, by4, bx4,
 1037|   775k|                                           have_top, have_left);
 1038|   775k|            b->intra = !dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   775k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1039|   775k|                            ts->cdf.m.intra[ictx]);
 1040|   775k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   775k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 775k]
  |  |  ------------------
  |  |   35|   775k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   775k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1041|      0|                printf("Post-intra[%d]: r=%d\n", b->intra, ts->msac.rng);
 1042|   775k|        }
 1043|  2.93M|    } else if (f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (1043:16): [True: 2.28M, False: 650k]
  ------------------
 1044|  2.28M|        b->intra = !dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.intrabc);
  ------------------
  |  |   52|  2.28M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1045|  2.28M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.28M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.28M]
  |  |  ------------------
  |  |   35|  2.28M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.28M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1046|      0|            printf("Post-intrabcflag[%d]: r=%d\n", b->intra, ts->msac.rng);
 1047|  2.28M|    } else {
 1048|   650k|        b->intra = 1;
 1049|   650k|    }
 1050|       |
 1051|       |    // intra/inter-specific stuff
 1052|  4.10M|    if (b->intra) {
  ------------------
  |  Branch (1052:9): [True: 2.37M, False: 1.73M]
  ------------------
 1053|  2.37M|        uint16_t *const ymode_cdf = IS_INTER_OR_SWITCH(f->frame_hdr) ?
  ------------------
  |  |   36|  2.37M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 119k, False: 2.25M]
  |  |  ------------------
  ------------------
 1054|   119k|            ts->cdf.m.y_mode[dav1d_ymode_size_context[bs]] :
 1055|  2.37M|            ts->cdf.kfym[dav1d_intra_mode_context[t->a->mode[bx4]]]
 1056|  2.25M|                        [dav1d_intra_mode_context[t->l.mode[by4]]];
 1057|  2.37M|        b->y_mode = dav1d_msac_decode_symbol_adapt16(&ts->msac, ymode_cdf,
  ------------------
  |  |   57|  2.37M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1058|  2.37M|                                                     N_INTRA_PRED_MODES - 1);
 1059|  2.37M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.37M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.37M]
  |  |  ------------------
  |  |   35|  2.37M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.37M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1060|      0|            printf("Post-ymode[%d]: r=%d\n", b->y_mode, ts->msac.rng);
 1061|       |
 1062|       |        // angle delta
 1063|  2.37M|        if (b_dim[2] + b_dim[3] >= 2 && b->y_mode >= VERT_PRED &&
  ------------------
  |  Branch (1063:13): [True: 1.93M, False: 436k]
  |  Branch (1063:41): [True: 990k, False: 946k]
  ------------------
 1064|   990k|            b->y_mode <= VERT_LEFT_PRED)
  ------------------
  |  Branch (1064:13): [True: 524k, False: 465k]
  ------------------
 1065|   524k|        {
 1066|   524k|            uint16_t *const acdf = ts->cdf.m.angle_delta[b->y_mode - VERT_PRED];
 1067|   524k|            const int angle = dav1d_msac_decode_symbol_adapt8(&ts->msac, acdf, 6);
  ------------------
  |  |   48|   524k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1068|   524k|            b->y_angle = angle - 3;
 1069|  1.84M|        } else {
 1070|  1.84M|            b->y_angle = 0;
 1071|  1.84M|        }
 1072|       |
 1073|  2.37M|        if (has_chroma) {
  ------------------
  |  Branch (1073:13): [True: 1.66M, False: 706k]
  ------------------
 1074|  1.66M|            const int cfl_allowed = f->frame_hdr->segmentation.lossless[b->seg_id] ?
  ------------------
  |  Branch (1074:37): [True: 38.7k, False: 1.62M]
  ------------------
 1075|  1.62M|                cbw4 == 1 && cbh4 == 1 : !!(cfl_allowed_mask & (1 << bs));
  ------------------
  |  Branch (1075:17): [True: 21.2k, False: 17.5k]
  |  Branch (1075:30): [True: 17.0k, False: 4.28k]
  ------------------
 1076|  1.66M|            uint16_t *const uvmode_cdf = ts->cdf.m.uv_mode[cfl_allowed][b->y_mode];
 1077|  1.66M|            b->uv_mode = dav1d_msac_decode_symbol_adapt16(&ts->msac, uvmode_cdf,
  ------------------
  |  |   57|  1.66M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1078|  1.66M|                             N_UV_INTRA_PRED_MODES - 1 - !cfl_allowed);
 1079|  1.66M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.66M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.66M]
  |  |  ------------------
  |  |   35|  1.66M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.66M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1080|      0|                printf("Post-uvmode[%d]: r=%d\n", b->uv_mode, ts->msac.rng);
 1081|       |
 1082|  1.66M|            b->uv_angle = 0;
 1083|  1.66M|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1083:17): [True: 351k, False: 1.31M]
  ------------------
 1084|   351k|#define SIGN(a) (!!(a) + ((a) > 0))
 1085|   351k|                const int sign = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   351k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1086|   351k|                                     ts->cdf.m.cfl_sign, 7) + 1;
 1087|   351k|                const int sign_u = sign * 0x56 >> 8, sign_v = sign - sign_u * 3;
 1088|   351k|                assert(sign_u == sign / 3);
  ------------------
  |  Branch (1088:17): [True: 351k, False: 0]
  ------------------
 1089|   351k|                if (sign_u) {
  ------------------
  |  Branch (1089:21): [True: 328k, False: 23.0k]
  ------------------
 1090|   328k|                    const int ctx = (sign_u == 2) * 3 + sign_v;
 1091|   328k|                    b->cfl_alpha[0] = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   328k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1092|   328k|                                          ts->cdf.m.cfl_alpha[ctx], 15) + 1;
 1093|   328k|                    if (sign_u == 1) b->cfl_alpha[0] = -b->cfl_alpha[0];
  ------------------
  |  Branch (1093:25): [True: 224k, False: 104k]
  ------------------
 1094|   328k|                } else {
 1095|  23.0k|                    b->cfl_alpha[0] = 0;
 1096|  23.0k|                }
 1097|   351k|                if (sign_v) {
  ------------------
  |  Branch (1097:21): [True: 240k, False: 111k]
  ------------------
 1098|   240k|                    const int ctx = (sign_v == 2) * 3 + sign_u;
 1099|   240k|                    b->cfl_alpha[1] = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   240k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1100|   240k|                                          ts->cdf.m.cfl_alpha[ctx], 15) + 1;
 1101|   240k|                    if (sign_v == 1) b->cfl_alpha[1] = -b->cfl_alpha[1];
  ------------------
  |  Branch (1101:25): [True: 97.6k, False: 142k]
  ------------------
 1102|   240k|                } else {
 1103|   111k|                    b->cfl_alpha[1] = 0;
 1104|   111k|                }
 1105|   351k|#undef SIGN
 1106|   351k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   351k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 351k]
  |  |  ------------------
  |  |   35|   351k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   351k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1107|      0|                    printf("Post-uvalphas[%d/%d]: r=%d\n",
 1108|      0|                           b->cfl_alpha[0], b->cfl_alpha[1], ts->msac.rng);
 1109|  1.31M|            } else if (b_dim[2] + b_dim[3] >= 2 && b->uv_mode >= VERT_PRED &&
  ------------------
  |  Branch (1109:24): [True: 1.11M, False: 205k]
  |  Branch (1109:52): [True: 650k, False: 459k]
  ------------------
 1110|   650k|                       b->uv_mode <= VERT_LEFT_PRED)
  ------------------
  |  Branch (1110:24): [True: 323k, False: 327k]
  ------------------
 1111|   323k|            {
 1112|   323k|                uint16_t *const acdf = ts->cdf.m.angle_delta[b->uv_mode - VERT_PRED];
 1113|   323k|                const int angle = dav1d_msac_decode_symbol_adapt8(&ts->msac, acdf, 6);
  ------------------
  |  |   48|   323k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1114|   323k|                b->uv_angle = angle - 3;
 1115|   323k|            }
 1116|  1.66M|        }
 1117|       |
 1118|  2.37M|        b->pal_sz[0] = b->pal_sz[1] = 0;
 1119|  2.37M|        if (f->frame_hdr->allow_screen_content_tools &&
  ------------------
  |  Branch (1119:13): [True: 1.77M, False: 594k]
  ------------------
 1120|  1.77M|            imax(bw4, bh4) <= 16 && bw4 + bh4 >= 4)
  ------------------
  |  Branch (1120:13): [True: 1.69M, False: 86.4k]
  |  Branch (1120:37): [True: 1.38M, False: 303k]
  ------------------
 1121|  1.38M|        {
 1122|  1.38M|            const int sz_ctx = b_dim[2] + b_dim[3] - 2;
 1123|  1.38M|            if (b->y_mode == DC_PRED) {
  ------------------
  |  Branch (1123:17): [True: 683k, False: 705k]
  ------------------
 1124|   683k|                const int pal_ctx = (t->a->pal_sz[bx4] > 0) + (t->l.pal_sz[by4] > 0);
 1125|   683k|                const int use_y_pal = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   683k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1126|   683k|                                          ts->cdf.m.pal_y[sz_ctx][pal_ctx]);
 1127|   683k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   683k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 683k]
  |  |  ------------------
  |  |   35|   683k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   683k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1128|      0|                    printf("Post-y_pal[%d]: r=%d\n", use_y_pal, ts->msac.rng);
 1129|   683k|                if (use_y_pal)
  ------------------
  |  Branch (1129:21): [True: 79.5k, False: 604k]
  ------------------
 1130|  79.5k|                    f->bd_fn.read_pal_plane(t, b, 0, sz_ctx, bx4, by4);
 1131|   683k|            }
 1132|       |
 1133|  1.38M|            if (has_chroma && b->uv_mode == DC_PRED) {
  ------------------
  |  Branch (1133:17): [True: 1.07M, False: 316k]
  |  Branch (1133:31): [True: 326k, False: 745k]
  ------------------
 1134|   326k|                const int pal_ctx = b->pal_sz[0] > 0;
 1135|   326k|                const int use_uv_pal = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   326k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1136|   326k|                                           ts->cdf.m.pal_uv[pal_ctx]);
 1137|   326k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   326k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 326k]
  |  |  ------------------
  |  |   35|   326k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   326k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1138|      0|                    printf("Post-uv_pal[%d]: r=%d\n", use_uv_pal, ts->msac.rng);
 1139|   326k|                if (use_uv_pal) // see aomedia bug 2183 for why we use luma coordinates
  ------------------
  |  Branch (1139:21): [True: 17.1k, False: 309k]
  ------------------
 1140|  17.1k|                    f->bd_fn.read_pal_uv(t, b, sz_ctx, bx4, by4);
 1141|   326k|            }
 1142|  1.38M|        }
 1143|       |
 1144|  2.37M|        if (b->y_mode == DC_PRED && !b->pal_sz[0] &&
  ------------------
  |  Branch (1144:13): [True: 1.11M, False: 1.26M]
  |  Branch (1144:37): [True: 1.03M, False: 79.5k]
  ------------------
 1145|  1.03M|            imax(b_dim[2], b_dim[3]) <= 3 && f->seq_hdr->filter_intra)
  ------------------
  |  Branch (1145:13): [True: 782k, False: 249k]
  |  Branch (1145:46): [True: 485k, False: 297k]
  ------------------
 1146|   485k|        {
 1147|   485k|            const int is_filter = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   485k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1148|   485k|                                      ts->cdf.m.use_filter_intra[bs]);
 1149|   485k|            if (is_filter) {
  ------------------
  |  Branch (1149:17): [True: 317k, False: 168k]
  ------------------
 1150|   317k|                b->y_mode = FILTER_PRED;
 1151|   317k|                b->y_angle = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   317k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1152|   317k|                                 ts->cdf.m.filter_intra, 4);
 1153|   317k|            }
 1154|   485k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   485k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 485k]
  |  |  ------------------
  |  |   35|   485k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   485k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1155|      0|                printf("Post-filterintramode[%d/%d]: r=%d\n",
 1156|      0|                       b->y_mode, b->y_angle, ts->msac.rng);
 1157|   485k|        }
 1158|       |
 1159|  2.37M|        if (b->pal_sz[0]) {
  ------------------
  |  Branch (1159:13): [True: 79.5k, False: 2.29M]
  ------------------
 1160|  79.5k|            uint8_t *pal_idx;
 1161|  79.5k|            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1161:17): [True: 0, False: 79.5k]
  ------------------
 1162|      0|                const int p = t->frame_thread.pass & 1;
 1163|      0|                assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1163:17): [True: 0, False: 0]
  ------------------
 1164|      0|                pal_idx = ts->frame_thread[p].pal_idx;
 1165|      0|                ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1166|      0|            } else
 1167|  79.5k|                pal_idx = t->scratch.pal_idx_y;
 1168|  79.5k|            read_pal_indices(t, pal_idx, b->pal_sz[0], 0, w4, h4, bw4, bh4);
 1169|  79.5k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  79.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 79.5k]
  |  |  ------------------
  |  |   35|  79.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  79.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1170|      0|                printf("Post-y-pal-indices: r=%d\n", ts->msac.rng);
 1171|  79.5k|        }
 1172|       |
 1173|  2.37M|        if (has_chroma && b->pal_sz[1]) {
  ------------------
  |  Branch (1173:13): [True: 1.66M, False: 706k]
  |  Branch (1173:27): [True: 17.1k, False: 1.65M]
  ------------------
 1174|  17.1k|            uint8_t *pal_idx;
 1175|  17.1k|            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1175:17): [True: 0, False: 17.1k]
  ------------------
 1176|      0|                const int p = t->frame_thread.pass & 1;
 1177|      0|                assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1177:17): [True: 0, False: 0]
  ------------------
 1178|      0|                pal_idx = ts->frame_thread[p].pal_idx;
 1179|      0|                ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1180|      0|            } else
 1181|  17.1k|                pal_idx = t->scratch.pal_idx_uv;
 1182|  17.1k|            read_pal_indices(t, pal_idx, b->pal_sz[1], 1, cw4, ch4, cbw4, cbh4);
 1183|  17.1k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  17.1k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 17.1k]
  |  |  ------------------
  |  |   35|  17.1k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  17.1k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1184|      0|                printf("Post-uv-pal-indices: r=%d\n", ts->msac.rng);
 1185|  17.1k|        }
 1186|       |
 1187|  2.37M|        const TxfmInfo *t_dim;
 1188|  2.37M|        if (f->frame_hdr->segmentation.lossless[b->seg_id]) {
  ------------------
  |  Branch (1188:13): [True: 65.3k, False: 2.30M]
  ------------------
 1189|  65.3k|            b->tx = b->uvtx = (int) TX_4X4;
 1190|  65.3k|            t_dim = &dav1d_txfm_dimensions[TX_4X4];
 1191|  2.30M|        } else {
 1192|  2.30M|            b->tx = dav1d_max_txfm_size_for_bs[bs][0];
 1193|  2.30M|            b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
 1194|  2.30M|            t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|  2.30M|            if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE && t_dim->max > TX_4X4) {
  ------------------
  |  Branch (1195:17): [True: 651k, False: 1.65M]
  |  Branch (1195:67): [True: 590k, False: 61.0k]
  ------------------
 1196|   590k|                const int tctx = get_tx_ctx(t->a, &t->l, t_dim, by4, bx4);
 1197|   590k|                uint16_t *const tx_cdf = ts->cdf.m.txsz[t_dim->max - 1][tctx];
 1198|   590k|                int depth = dav1d_msac_decode_symbol_adapt4(&ts->msac, tx_cdf,
  ------------------
  |  |   47|   590k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1199|   590k|                                imin(t_dim->max, 2));
 1200|       |
 1201|  1.13M|                while (depth--) {
  ------------------
  |  Branch (1201:24): [True: 542k, False: 590k]
  ------------------
 1202|   542k|                    b->tx = t_dim->sub;
 1203|   542k|                    t_dim = &dav1d_txfm_dimensions[b->tx];
 1204|   542k|                }
 1205|   590k|            }
 1206|  2.30M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.30M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.30M]
  |  |  ------------------
  |  |   35|  2.30M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.30M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1207|      0|                printf("Post-tx[%d]: r=%d\n", b->tx, ts->msac.rng);
 1208|  2.30M|        }
 1209|       |
 1210|       |        // reconstruction
 1211|  2.37M|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1211:13): [True: 0, False: 2.37M]
  ------------------
 1212|      0|            f->bd_fn.read_coef_blocks(t, bs, b);
 1213|  2.37M|        } else {
 1214|  2.37M|            f->bd_fn.recon_b_intra(t, bs, intra_edge_flags, b);
 1215|  2.37M|        }
 1216|       |
 1217|  2.37M|        if (f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (1217:13): [True: 383k, False: 1.98M]
  ------------------
 1218|  1.98M|            f->frame_hdr->loopfilter.level_y[1])
  ------------------
  |  Branch (1218:13): [True: 119k, False: 1.87M]
  ------------------
 1219|   503k|        {
 1220|   503k|            dav1d_create_lf_mask_intra(t->lf_mask, f->lf.level, f->b4_stride,
 1221|   503k|                                       (const uint8_t (*)[8][2])
 1222|   503k|                                       &ts->lflvl[b->seg_id][0][0][0],
 1223|   503k|                                       t->bx, t->by, f->w4, f->h4, bs,
 1224|   503k|                                       b->tx, b->uvtx, f->cur.p.layout,
 1225|   503k|                                       &t->a->tx_lpf_y[bx4], &t->l.tx_lpf_y[by4],
 1226|   503k|                                       has_chroma ? &t->a->tx_lpf_uv[cbx4] : NULL,
  ------------------
  |  Branch (1226:40): [True: 356k, False: 146k]
  ------------------
 1227|   503k|                                       has_chroma ? &t->l.tx_lpf_uv[cby4] : NULL);
  ------------------
  |  Branch (1227:40): [True: 356k, False: 146k]
  ------------------
 1228|   503k|        }
 1229|       |        // update contexts
 1230|  2.37M|        const enum IntraPredMode y_mode_nofilt =
 1231|  2.37M|            b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
  ------------------
  |  Branch (1231:13): [True: 317k, False: 2.05M]
  ------------------
 1232|  2.37M|        BlockContext *edge = t->a;
 1233|  7.12M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1233:36): [True: 4.74M, False: 2.37M]
  ------------------
 1234|  4.74M|            int t_lsz = ((uint8_t *) &t_dim->lw)[i]; // lw then lh
 1235|  4.74M|#define set_ctx(rep_macro) \
 1236|  4.74M|            rep_macro(edge->tx_intra, off, t_lsz); \
 1237|  4.74M|            rep_macro(edge->tx, off, t_lsz); \
 1238|  4.74M|            rep_macro(edge->mode, off, y_mode_nofilt); \
 1239|  4.74M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
 1240|  4.74M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1241|  4.74M|            rep_macro(edge->skip_mode, off, 0); \
 1242|  4.74M|            rep_macro(edge->intra, off, 1); \
 1243|  4.74M|            rep_macro(edge->skip, off, b->skip); \
 1244|       |            /* see aomedia bug 2183 for why we use luma coordinates here */ \
 1245|  4.74M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
 1246|  4.74M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
 1247|  4.74M|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
 1248|  4.74M|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
 1249|  4.74M|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
 1250|  4.74M|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
 1251|  4.74M|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
 1252|  4.74M|            }
 1253|  4.74M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  4.74M|    switch (var) { \
  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1236|   913k|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|   913k|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|   913k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|   913k|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|   913k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|   913k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|   913k|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|   913k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   913k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|   913k|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|   913k|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   913k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  1.82M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (56:43): [True: 616k, False: 296k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|   913k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|   913k|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 62.8k, False: 850k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  62.8k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  62.8k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  62.8k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  62.8k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  62.8k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  62.8k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  62.8k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  62.8k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  62.8k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  62.8k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  62.8k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  62.8k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  62.8k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  62.8k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  62.8k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  62.8k|            }
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 913k, False: 3.83M]
  |  |  ------------------
  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1236|  1.33M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  1.33M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  1.33M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  1.33M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  1.33M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  1.33M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  1.33M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  1.33M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.33M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  1.33M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  1.33M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.33M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  2.66M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (58:45): [True: 977k, False: 357k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  1.33M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  1.33M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 84.1k, False: 1.25M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  84.1k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  84.1k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  84.1k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  84.1k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  84.1k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  84.1k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  84.1k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  84.1k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  84.1k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  84.1k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  84.1k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  84.1k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  84.1k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  84.1k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  84.1k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  84.1k|            }
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 1.33M, False: 3.41M]
  |  |  ------------------
  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1236|  1.22M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  1.22M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  1.22M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  1.22M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  1.22M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  1.22M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  1.22M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  1.22M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.22M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  1.22M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  1.22M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.22M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  2.44M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (60:45): [True: 935k, False: 288k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  1.22M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  1.22M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 57.5k, False: 1.16M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  57.5k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  57.5k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  57.5k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  57.5k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  57.5k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  57.5k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  57.5k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  57.5k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  57.5k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  57.5k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  57.5k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  57.5k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  57.5k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  57.5k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  57.5k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  57.5k|            }
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 1.22M, False: 3.52M]
  |  |  ------------------
  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1236|   667k|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|   667k|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|   667k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|   667k|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|   667k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|   667k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|   667k|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|   667k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   667k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|   667k|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|   667k|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   667k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.33M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (62:45): [True: 468k, False: 199k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|   667k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|   667k|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 19.6k, False: 648k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  19.6k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  19.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  19.6k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  19.6k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  19.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  19.6k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  19.6k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  19.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  19.6k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  19.6k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  19.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  19.6k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  19.6k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  19.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  19.6k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  19.6k|            }
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 667k, False: 4.07M]
  |  |  ------------------
  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1236|   417k|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|   417k|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|   417k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|   417k|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|   417k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|   417k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|   417k|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|   417k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   417k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|   417k|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|   417k|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   417k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   417k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   834k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (64:29): [True: 224k, False: 192k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   65|   417k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 417k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|   417k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|   417k|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 7.90k, False: 409k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  7.90k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  7.90k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  7.90k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  7.90k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  7.90k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 7.90k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  7.90k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  7.90k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  7.90k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  7.90k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  7.90k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 7.90k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  7.90k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  7.90k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  7.90k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  7.90k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  7.90k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 7.90k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  7.90k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  7.90k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  7.90k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  7.90k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  7.90k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 7.90k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  7.90k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  7.90k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  7.90k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  7.90k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  7.90k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 7.90k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  7.90k|            }
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 417k, False: 4.32M]
  |  |  ------------------
  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1236|   190k|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|   190k|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|   190k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|   190k|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|   190k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|   190k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|   190k|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|   190k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   190k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|   190k|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|   190k|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   190k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   190k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   380k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (67:29): [True: 112k, False: 77.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   68|   190k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 190k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|   190k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|   190k|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 6.80k, False: 183k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  6.80k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.80k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.80k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.80k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.80k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.80k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  6.80k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.80k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.80k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.80k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.80k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.80k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  6.80k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.80k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.80k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.80k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.80k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.80k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  6.80k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.80k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.80k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.80k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.80k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.80k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  6.80k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.80k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.80k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.80k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.80k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.80k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  6.80k|            }
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 190k, False: 4.55M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 4.74M]
  |  |  ------------------
  |  |   78|  4.74M|    }
  ------------------
  |  Branch (1253:13): [Folded, False: 0]
  ------------------
 1254|  4.74M|#undef set_ctx
 1255|  4.74M|        }
 1256|  2.37M|        if (b->pal_sz[0])
  ------------------
  |  Branch (1256:13): [True: 79.5k, False: 2.29M]
  ------------------
 1257|  79.5k|            f->bd_fn.copy_pal_block_y(t, bx4, by4, bw4, bh4);
 1258|  2.37M|        if (has_chroma) {
  ------------------
  |  Branch (1258:13): [True: 1.66M, False: 706k]
  ------------------
 1259|  1.66M|            uint8_t uv_mode = b->uv_mode;
 1260|  1.66M|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
 1261|  1.66M|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
 1262|  1.66M|            if (b->pal_sz[1])
  ------------------
  |  Branch (1262:17): [True: 17.1k, False: 1.65M]
  ------------------
 1263|  17.1k|                f->bd_fn.copy_pal_block_uv(t, bx4, by4, bw4, bh4);
 1264|  1.66M|        }
 1265|  2.37M|        if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc)
  ------------------
  |  |   36|  4.74M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 119k, False: 2.25M]
  |  |  ------------------
  ------------------
  |  Branch (1265:49): [True: 1.60M, False: 650k]
  ------------------
 1266|  1.72M|            splat_intraref(f->c, t, bs, bw4, bh4);
 1267|  2.37M|    } else if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  1.73M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  1.73M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 676k, False: 1.05M]
  |  |  ------------------
  ------------------
 1268|       |        // intra block copy
 1269|   676k|        refmvs_candidate mvstack[8];
 1270|   676k|        int n_mvs, ctx;
 1271|   676k|        dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1272|   676k|                          (union refmvs_refpair) { .ref = { 0, -1 }},
 1273|   676k|                          bs, intra_edge_flags, t->by, t->bx);
 1274|       |
 1275|   676k|        if (mvstack[0].mv.mv[0].n)
  ------------------
  |  Branch (1275:13): [True: 619k, False: 57.8k]
  ------------------
 1276|   619k|            b->mv[0] = mvstack[0].mv.mv[0];
 1277|  57.8k|        else if (mvstack[1].mv.mv[0].n)
  ------------------
  |  Branch (1277:18): [True: 0, False: 57.8k]
  ------------------
 1278|      0|            b->mv[0] = mvstack[1].mv.mv[0];
 1279|  57.8k|        else {
 1280|  57.8k|            if (t->by - (16 << f->seq_hdr->sb128) < ts->tiling.row_start) {
  ------------------
  |  Branch (1280:17): [True: 56.9k, False: 930]
  ------------------
 1281|  56.9k|                b->mv[0].y = 0;
 1282|  56.9k|                b->mv[0].x = -(512 << f->seq_hdr->sb128) - 2048;
 1283|  56.9k|            } else {
 1284|    930|                b->mv[0].y = -(512 << f->seq_hdr->sb128);
 1285|    930|                b->mv[0].x = 0;
 1286|    930|            }
 1287|  57.8k|        }
 1288|       |
 1289|   676k|        const union mv ref = b->mv[0];
 1290|   676k|        read_mv_residual(ts, &b->mv[0], -1);
 1291|       |
 1292|       |        // clip intrabc motion vector to decoded parts of current tile
 1293|   676k|        int border_left = ts->tiling.col_start * 4;
 1294|   676k|        int border_top  = ts->tiling.row_start * 4;
 1295|   676k|        if (has_chroma) {
  ------------------
  |  Branch (1295:13): [True: 274k, False: 402k]
  ------------------
 1296|   274k|            if (bw4 < 2 &&  ss_hor)
  ------------------
  |  Branch (1296:17): [True: 95.7k, False: 178k]
  |  Branch (1296:29): [True: 8.58k, False: 87.1k]
  ------------------
 1297|  8.58k|                border_left += 4;
 1298|   274k|            if (bh4 < 2 &&  ss_ver)
  ------------------
  |  Branch (1298:17): [True: 86.4k, False: 188k]
  |  Branch (1298:29): [True: 6.06k, False: 80.3k]
  ------------------
 1299|  6.06k|                border_top  += 4;
 1300|   274k|        }
 1301|   676k|        int src_left   = t->bx * 4 + (b->mv[0].x >> 3);
 1302|   676k|        int src_top    = t->by * 4 + (b->mv[0].y >> 3);
 1303|   676k|        int src_right  = src_left + bw4 * 4;
 1304|   676k|        int src_bottom = src_top  + bh4 * 4;
 1305|   676k|        const int border_right = ((ts->tiling.col_end + (bw4 - 1)) & ~(bw4 - 1)) * 4;
 1306|       |
 1307|       |        // check against left or right tile boundary and adjust if necessary
 1308|   676k|        if (src_left < border_left) {
  ------------------
  |  Branch (1308:13): [True: 214k, False: 462k]
  ------------------
 1309|   214k|            src_right += border_left - src_left;
 1310|   214k|            src_left  += border_left - src_left;
 1311|   462k|        } else if (src_right > border_right) {
  ------------------
  |  Branch (1311:20): [True: 230k, False: 232k]
  ------------------
 1312|   230k|            src_left  -= src_right - border_right;
 1313|   230k|            src_right -= src_right - border_right;
 1314|   230k|        }
 1315|       |        // check against top tile boundary and adjust if necessary
 1316|   676k|        if (src_top < border_top) {
  ------------------
  |  Branch (1316:13): [True: 578k, False: 98.5k]
  ------------------
 1317|   578k|            src_bottom += border_top - src_top;
 1318|   578k|            src_top    += border_top - src_top;
 1319|   578k|        }
 1320|       |
 1321|   676k|        const int sbx = (t->bx >> (4 + f->seq_hdr->sb128)) << (6 + f->seq_hdr->sb128);
 1322|   676k|        const int sby = (t->by >> (4 + f->seq_hdr->sb128)) << (6 + f->seq_hdr->sb128);
 1323|   676k|        const int sb_size = 1 << (6 + f->seq_hdr->sb128);
 1324|       |        // check for overlap with current superblock
 1325|   676k|        if (src_bottom > sby && src_right > sbx) {
  ------------------
  |  Branch (1325:13): [True: 651k, False: 25.4k]
  |  Branch (1325:33): [True: 238k, False: 412k]
  ------------------
 1326|   238k|            if (src_top - border_top >= src_bottom - sby) {
  ------------------
  |  Branch (1326:17): [True: 1.07k, False: 237k]
  ------------------
 1327|       |                // if possible move src up into the previous suberblock row
 1328|  1.07k|                src_top    -= src_bottom - sby;
 1329|  1.07k|                src_bottom -= src_bottom - sby;
 1330|   237k|            } else if (src_left - border_left >= src_right - sbx) {
  ------------------
  |  Branch (1330:24): [True: 229k, False: 7.77k]
  ------------------
 1331|       |                // if possible move src left into the previous suberblock
 1332|   229k|                src_left  -= src_right - sbx;
 1333|   229k|                src_right -= src_right - sbx;
 1334|   229k|            }
 1335|   238k|        }
 1336|       |        // move src up if it is below current superblock row
 1337|   676k|        if (src_bottom > sby + sb_size) {
  ------------------
  |  Branch (1337:13): [True: 4.27k, False: 672k]
  ------------------
 1338|  4.27k|            src_top    -= src_bottom - (sby + sb_size);
 1339|  4.27k|            src_bottom -= src_bottom - (sby + sb_size);
 1340|  4.27k|        }
 1341|       |        // error out if mv still overlaps with the current superblock
 1342|   676k|        if (src_bottom > sby && src_right > sbx)
  ------------------
  |  Branch (1342:13): [True: 650k, False: 26.5k]
  |  Branch (1342:33): [True: 7.77k, False: 642k]
  ------------------
 1343|  7.77k|            return -1;
 1344|       |
 1345|   669k|        b->mv[0].x = (src_left - t->bx * 4) * 8;
 1346|   669k|        b->mv[0].y = (src_top  - t->by * 4) * 8;
 1347|       |
 1348|   669k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   669k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 669k]
  |  |  ------------------
  |  |   35|   669k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   669k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1349|      0|            printf("Post-dmv[%d/%d,ref=%d/%d|%d/%d]: r=%d\n",
 1350|      0|                   b->mv[0].y, b->mv[0].x, ref.y, ref.x,
 1351|      0|                   mvstack[0].mv.mv[0].y, mvstack[0].mv.mv[0].x, ts->msac.rng);
 1352|   669k|        read_vartx_tree(t, b, bs, bx4, by4);
 1353|       |
 1354|       |        // reconstruction
 1355|   669k|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1355:13): [True: 0, False: 669k]
  ------------------
 1356|      0|            f->bd_fn.read_coef_blocks(t, bs, b);
 1357|      0|            b->filter2d = FILTER_2D_BILINEAR;
 1358|   669k|        } else {
 1359|   669k|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (1359:17): [True: 0, False: 669k]
  ------------------
 1360|   669k|        }
 1361|       |
 1362|   669k|        splat_intrabc_mv(f->c, t, bs, b, bw4, bh4);
 1363|   669k|        BlockContext *edge = t->a;
 1364|  2.00M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1364:36): [True: 1.33M, False: 669k]
  ------------------
 1365|  1.33M|#define set_ctx(rep_macro) \
 1366|  1.33M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
 1367|  1.33M|            rep_macro(edge->mode, off, DC_PRED); \
 1368|  1.33M|            rep_macro(edge->pal_sz, off, 0); \
 1369|       |            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
 1370|  1.33M|            rep_macro(t->pal_sz_uv[i], off, 0); \
 1371|  1.33M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1372|  1.33M|            rep_macro(edge->skip_mode, off, 0); \
 1373|  1.33M|            rep_macro(edge->intra, off, 0); \
 1374|  1.33M|            rep_macro(edge->skip, off, b->skip)
 1375|  1.33M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  1.33M|    switch (var) { \
  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1366|   791k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   791k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   791k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   791k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   791k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   791k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   791k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   791k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   791k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   791k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   791k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 791k, False: 547k]
  |  |  ------------------
  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1366|   146k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   146k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   146k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   146k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   146k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   146k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   146k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   146k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   146k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   146k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   146k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 146k, False: 1.19M]
  |  |  ------------------
  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1366|   187k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   187k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   187k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   187k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   187k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   187k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   187k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   187k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   187k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   187k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   187k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 187k, False: 1.15M]
  |  |  ------------------
  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1366|   100k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   100k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   100k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   100k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   100k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   100k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   100k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   100k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   100k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   100k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 100k, False: 1.23M]
  |  |  ------------------
  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1366|   106k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   106k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   106k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   106k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   106k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   106k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   106k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   106k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   106k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   106k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   106k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   106k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   106k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 106k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 106k, False: 1.23M]
  |  |  ------------------
  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1366|  6.77k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|  6.77k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|  6.77k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|  6.77k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|  6.77k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|  6.77k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|  6.77k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|  6.77k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|  6.77k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  6.77k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  6.77k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  6.77k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  6.77k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 6.77k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 6.77k, False: 1.33M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 1.33M]
  |  |  ------------------
  |  |   78|  1.33M|    }
  ------------------
  |  Branch (1375:13): [Folded, False: 0]
  ------------------
 1376|  1.33M|#undef set_ctx
 1377|  1.33M|        }
 1378|   669k|        if (has_chroma) {
  ------------------
  |  Branch (1378:13): [True: 268k, False: 400k]
  ------------------
 1379|   268k|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
 1380|   268k|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
 1381|   268k|        }
 1382|  1.05M|    } else {
 1383|       |        // inter-specific mode/mv coding
 1384|  1.05M|        int is_comp, has_subpel_filter;
 1385|       |
 1386|  1.05M|        if (b->skip_mode) {
  ------------------
  |  Branch (1386:13): [True: 14.8k, False: 1.04M]
  ------------------
 1387|  14.8k|            is_comp = 1;
 1388|  1.04M|        } else if ((!seg || (seg->ref == -1 && !seg->globalmv && !seg->skip)) &&
  ------------------
  |  Branch (1388:21): [True: 561k, False: 483k]
  |  Branch (1388:30): [True: 154k, False: 328k]
  |  Branch (1388:48): [True: 98.1k, False: 56.4k]
  |  Branch (1388:66): [True: 78.3k, False: 19.8k]
  ------------------
 1389|   639k|                   f->frame_hdr->switchable_comp_refs && imin(bw4, bh4) > 1)
  ------------------
  |  Branch (1389:20): [True: 407k, False: 232k]
  |  Branch (1389:58): [True: 276k, False: 130k]
  ------------------
 1390|   276k|        {
 1391|   276k|            const int ctx = get_comp_ctx(t->a, &t->l, by4, bx4,
 1392|   276k|                                         have_top, have_left);
 1393|   276k|            is_comp = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   276k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1394|   276k|                          ts->cdf.m.comp[ctx]);
 1395|   276k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   276k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 276k]
  |  |  ------------------
  |  |   35|   276k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   276k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1396|      0|                printf("Post-compflag[%d]: r=%d\n", is_comp, ts->msac.rng);
 1397|   767k|        } else {
 1398|   767k|            is_comp = 0;
 1399|   767k|        }
 1400|       |
 1401|  1.05M|        if (b->skip_mode) {
  ------------------
  |  Branch (1401:13): [True: 14.8k, False: 1.04M]
  ------------------
 1402|  14.8k|            b->ref[0] = f->frame_hdr->skip_mode_refs[0];
 1403|  14.8k|            b->ref[1] = f->frame_hdr->skip_mode_refs[1];
 1404|  14.8k|            b->comp_type = COMP_INTER_AVG;
 1405|  14.8k|            b->inter_mode = NEARESTMV_NEARESTMV;
 1406|  14.8k|            b->drl_idx = NEAREST_DRL;
 1407|  14.8k|            has_subpel_filter = 0;
 1408|       |
 1409|  14.8k|            refmvs_candidate mvstack[8];
 1410|  14.8k|            int n_mvs, ctx;
 1411|  14.8k|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1412|  14.8k|                              (union refmvs_refpair) { .ref = {
 1413|  14.8k|                                    b->ref[0] + 1, b->ref[1] + 1 }},
 1414|  14.8k|                              bs, intra_edge_flags, t->by, t->bx);
 1415|       |
 1416|  14.8k|            b->mv[0] = mvstack[0].mv.mv[0];
 1417|  14.8k|            b->mv[1] = mvstack[0].mv.mv[1];
 1418|  14.8k|            fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1419|  14.8k|            fix_mv_precision(f->frame_hdr, &b->mv[1]);
 1420|  14.8k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  14.8k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 14.8k]
  |  |  ------------------
  |  |   35|  14.8k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  14.8k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1421|      0|                printf("Post-skipmodeblock[mv=1:y=%d,x=%d,2:y=%d,x=%d,refs=%d+%d\n",
 1422|      0|                       b->mv[0].y, b->mv[0].x, b->mv[1].y, b->mv[1].x,
 1423|      0|                       b->ref[0], b->ref[1]);
 1424|  1.04M|        } else if (is_comp) {
  ------------------
  |  Branch (1424:20): [True: 154k, False: 890k]
  ------------------
 1425|   154k|            const int dir_ctx = get_comp_dir_ctx(t->a, &t->l, by4, bx4,
 1426|   154k|                                                 have_top, have_left);
 1427|   154k|            if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   154k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1427:17): [True: 130k, False: 23.4k]
  ------------------
 1428|   154k|                    ts->cdf.m.comp_dir[dir_ctx]))
 1429|   130k|            {
 1430|       |                // bidir - first reference (fw)
 1431|   130k|                const int ctx1 = av1_get_fwd_ref_ctx(t->a, &t->l, by4, bx4,
 1432|   130k|                                                     have_top, have_left);
 1433|   130k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   130k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1433:21): [True: 59.7k, False: 70.8k]
  ------------------
 1434|   130k|                        ts->cdf.m.comp_fwd_ref[0][ctx1]))
 1435|  59.7k|                {
 1436|  59.7k|                    const int ctx2 = av1_get_fwd_ref_2_ctx(t->a, &t->l, by4, bx4,
 1437|  59.7k|                                                           have_top, have_left);
 1438|  59.7k|                    b->ref[0] = 2 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  59.7k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1439|  59.7k|                                        ts->cdf.m.comp_fwd_ref[2][ctx2]);
 1440|  70.8k|                } else {
 1441|  70.8k|                    const int ctx2 = av1_get_fwd_ref_1_ctx(t->a, &t->l, by4, bx4,
 1442|  70.8k|                                                           have_top, have_left);
 1443|  70.8k|                    b->ref[0] = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  70.8k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1444|  70.8k|                                    ts->cdf.m.comp_fwd_ref[1][ctx2]);
 1445|  70.8k|                }
 1446|       |
 1447|       |                // second reference (bw)
 1448|   130k|                const int ctx3 = av1_get_bwd_ref_ctx(t->a, &t->l, by4, bx4,
 1449|   130k|                                                     have_top, have_left);
 1450|   130k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   130k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1450:21): [True: 68.8k, False: 61.8k]
  ------------------
 1451|   130k|                        ts->cdf.m.comp_bwd_ref[0][ctx3]))
 1452|  68.8k|                {
 1453|  68.8k|                    b->ref[1] = 6;
 1454|  68.8k|                } else {
 1455|  61.8k|                    const int ctx4 = av1_get_bwd_ref_1_ctx(t->a, &t->l, by4, bx4,
 1456|  61.8k|                                                           have_top, have_left);
 1457|  61.8k|                    b->ref[1] = 4 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  61.8k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1458|  61.8k|                                        ts->cdf.m.comp_bwd_ref[1][ctx4]);
 1459|  61.8k|                }
 1460|   130k|            } else {
 1461|       |                // unidir
 1462|  23.4k|                const int uctx_p = av1_get_uni_p_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  280|  23.4k|#define av1_get_uni_p_ctx av1_get_ref_ctx
  ------------------
 1463|  23.4k|                                                     have_top, have_left);
 1464|  23.4k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  23.4k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1464:21): [True: 4.79k, False: 18.6k]
  ------------------
 1465|  23.4k|                        ts->cdf.m.comp_uni_ref[0][uctx_p]))
 1466|  4.79k|                {
 1467|  4.79k|                    b->ref[0] = 4;
 1468|  4.79k|                    b->ref[1] = 6;
 1469|  18.6k|                } else {
 1470|  18.6k|                    const int uctx_p1 = av1_get_uni_p1_ctx(t->a, &t->l, by4, bx4,
 1471|  18.6k|                                                           have_top, have_left);
 1472|  18.6k|                    b->ref[0] = 0;
 1473|  18.6k|                    b->ref[1] = 1 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  18.6k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1474|  18.6k|                                        ts->cdf.m.comp_uni_ref[1][uctx_p1]);
 1475|  18.6k|                    if (b->ref[1] == 2) {
  ------------------
  |  Branch (1475:25): [True: 11.7k, False: 6.91k]
  ------------------
 1476|  11.7k|                        const int uctx_p2 = av1_get_uni_p2_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  281|  11.7k|#define av1_get_uni_p2_ctx av1_get_fwd_ref_2_ctx
  ------------------
 1477|  11.7k|                                                               have_top, have_left);
 1478|  11.7k|                        b->ref[1] += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  11.7k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1479|  11.7k|                                         ts->cdf.m.comp_uni_ref[2][uctx_p2]);
 1480|  11.7k|                    }
 1481|  18.6k|                }
 1482|  23.4k|            }
 1483|   154k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   154k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 154k]
  |  |  ------------------
  |  |   35|   154k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   154k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1484|      0|                printf("Post-refs[%d/%d]: r=%d\n",
 1485|      0|                       b->ref[0], b->ref[1], ts->msac.rng);
 1486|       |
 1487|   154k|            refmvs_candidate mvstack[8];
 1488|   154k|            int n_mvs, ctx;
 1489|   154k|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1490|   154k|                              (union refmvs_refpair) { .ref = {
 1491|   154k|                                    b->ref[0] + 1, b->ref[1] + 1 }},
 1492|   154k|                              bs, intra_edge_flags, t->by, t->bx);
 1493|       |
 1494|   154k|            b->inter_mode = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   154k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1495|   154k|                                ts->cdf.m.comp_inter_mode[ctx],
 1496|   154k|                                N_COMP_INTER_PRED_MODES - 1);
 1497|   154k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   154k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 154k]
  |  |  ------------------
  |  |   35|   154k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   154k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1498|      0|                printf("Post-compintermode[%d,ctx=%d,n_mvs=%d]: r=%d\n",
 1499|      0|                       b->inter_mode, ctx, n_mvs, ts->msac.rng);
 1500|       |
 1501|   154k|            const uint8_t *const im = dav1d_comp_inter_pred_modes[b->inter_mode];
 1502|   154k|            b->drl_idx = NEAREST_DRL;
 1503|   154k|            if (b->inter_mode == NEWMV_NEWMV) {
  ------------------
  |  Branch (1503:17): [True: 36.1k, False: 117k]
  ------------------
 1504|  36.1k|                if (n_mvs > 1) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1504:21): [True: 36.1k, False: 0]
  ------------------
 1505|  36.1k|                    const int drl_ctx_v1 = get_drl_context(mvstack, 0);
 1506|  36.1k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  36.1k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1507|  36.1k|                                      ts->cdf.m.drl_bit[drl_ctx_v1]);
 1508|  36.1k|                    if (b->drl_idx == NEARER_DRL && n_mvs > 2) {
  ------------------
  |  Branch (1508:25): [True: 26.6k, False: 9.47k]
  |  Branch (1508:53): [True: 8.23k, False: 18.4k]
  ------------------
 1509|  8.23k|                        const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1510|  8.23k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  8.23k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1511|  8.23k|                                          ts->cdf.m.drl_bit[drl_ctx_v2]);
 1512|  8.23k|                    }
 1513|  36.1k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  36.1k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 36.1k]
  |  |  ------------------
  |  |   35|  36.1k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  36.1k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1514|      0|                        printf("Post-drlidx[%d,n_mvs=%d]: r=%d\n",
 1515|      0|                               b->drl_idx, n_mvs, ts->msac.rng);
 1516|  36.1k|                }
 1517|   117k|            } else if (im[0] == NEARMV || im[1] == NEARMV) {
  ------------------
  |  Branch (1517:24): [True: 30.8k, False: 87.0k]
  |  Branch (1517:43): [True: 4.32k, False: 82.7k]
  ------------------
 1518|  35.1k|                b->drl_idx = NEARER_DRL;
 1519|  35.1k|                if (n_mvs > 2) { // NEAR or NEARISH
  ------------------
  |  Branch (1519:21): [True: 5.60k, False: 29.5k]
  ------------------
 1520|  5.60k|                    const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1521|  5.60k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  5.60k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1522|  5.60k|                                      ts->cdf.m.drl_bit[drl_ctx_v2]);
 1523|  5.60k|                    if (b->drl_idx == NEAR_DRL && n_mvs > 3) {
  ------------------
  |  Branch (1523:25): [True: 2.99k, False: 2.60k]
  |  Branch (1523:51): [True: 1.53k, False: 1.46k]
  ------------------
 1524|  1.53k|                        const int drl_ctx_v3 = get_drl_context(mvstack, 2);
 1525|  1.53k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.53k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1526|  1.53k|                                          ts->cdf.m.drl_bit[drl_ctx_v3]);
 1527|  1.53k|                    }
 1528|  5.60k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  5.60k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 5.60k]
  |  |  ------------------
  |  |   35|  5.60k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  5.60k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1529|      0|                        printf("Post-drlidx[%d,n_mvs=%d]: r=%d\n",
 1530|      0|                               b->drl_idx, n_mvs, ts->msac.rng);
 1531|  5.60k|                }
 1532|  35.1k|            }
 1533|   154k|            assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1533:13): [True: 154k, False: 0]
  |  Branch (1533:13): [True: 154k, False: 0]
  ------------------
 1534|       |
 1535|   154k|#define assign_comp_mv(idx) \
 1536|   154k|            switch (im[idx]) { \
 1537|   154k|            case NEARMV: \
 1538|   154k|            case NEARESTMV: \
 1539|   154k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
 1540|   154k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
 1541|   154k|                break; \
 1542|   154k|            case GLOBALMV: \
 1543|   154k|                has_subpel_filter |= \
 1544|   154k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
 1545|   154k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
 1546|   154k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
 1547|   154k|                break; \
 1548|   154k|            case NEWMV: \
 1549|   154k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
 1550|   154k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
 1551|   154k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
 1552|   154k|                break; \
 1553|   154k|            }
 1554|   154k|            has_subpel_filter = imin(bw4, bh4) == 1 ||
  ------------------
  |  Branch (1554:33): [True: 0, False: 154k]
  ------------------
 1555|   154k|                                b->inter_mode != GLOBALMV_GLOBALMV;
  ------------------
  |  Branch (1555:33): [True: 140k, False: 13.1k]
  ------------------
 1556|   154k|            assign_comp_mv(0);
  ------------------
  |  | 1536|   154k|            switch (im[idx]) { \
  |  |  ------------------
  |  |  |  Branch (1536:21): [True: 154k, False: 0]
  |  |  ------------------
  |  | 1537|  30.8k|            case NEARMV: \
  |  |  ------------------
  |  |  |  Branch (1537:13): [True: 30.8k, False: 123k]
  |  |  ------------------
  |  | 1538|  91.8k|            case NEARESTMV: \
  |  |  ------------------
  |  |  |  Branch (1538:13): [True: 61.0k, False: 92.9k]
  |  |  ------------------
  |  | 1539|  91.8k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1540|  91.8k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
  |  | 1541|  91.8k|                break; \
  |  | 1542|  30.8k|            case GLOBALMV: \
  |  |  ------------------
  |  |  |  Branch (1542:13): [True: 13.1k, False: 140k]
  |  |  ------------------
  |  | 1543|  13.1k|                has_subpel_filter |= \
  |  | 1544|  13.1k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
  |  | 1545|  13.1k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
  |  | 1546|  13.1k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
  |  | 1547|  13.1k|                break; \
  |  | 1548|  48.9k|            case NEWMV: \
  |  |  ------------------
  |  |  |  Branch (1548:13): [True: 48.9k, False: 105k]
  |  |  ------------------
  |  | 1549|  48.9k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1550|  48.9k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
  |  | 1551|  48.9k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
  |  | 1552|  48.9k|                break; \
  |  | 1553|   154k|            }
  ------------------
 1557|   154k|            assign_comp_mv(1);
  ------------------
  |  | 1536|   154k|            switch (im[idx]) { \
  |  |  ------------------
  |  |  |  Branch (1536:21): [True: 154k, False: 0]
  |  |  ------------------
  |  | 1537|  30.6k|            case NEARMV: \
  |  |  ------------------
  |  |  |  Branch (1537:13): [True: 30.6k, False: 123k]
  |  |  ------------------
  |  | 1538|  91.2k|            case NEARESTMV: \
  |  |  ------------------
  |  |  |  Branch (1538:13): [True: 60.6k, False: 93.4k]
  |  |  ------------------
  |  | 1539|  91.2k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1540|  91.2k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
  |  | 1541|  91.2k|                break; \
  |  | 1542|  30.6k|            case GLOBALMV: \
  |  |  ------------------
  |  |  |  Branch (1542:13): [True: 13.1k, False: 140k]
  |  |  ------------------
  |  | 1543|  13.1k|                has_subpel_filter |= \
  |  | 1544|  13.1k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
  |  | 1545|  13.1k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
  |  | 1546|  13.1k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
  |  | 1547|  13.1k|                break; \
  |  | 1548|  49.5k|            case NEWMV: \
  |  |  ------------------
  |  |  |  Branch (1548:13): [True: 49.5k, False: 104k]
  |  |  ------------------
  |  | 1549|  49.5k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1550|  49.5k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
  |  | 1551|  49.5k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
  |  | 1552|  49.5k|                break; \
  |  | 1553|   154k|            }
  ------------------
 1558|   154k|#undef assign_comp_mv
 1559|   154k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   154k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 154k]
  |  |  ------------------
  |  |   35|   154k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   154k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1560|      0|                printf("Post-residual_mv[1:y=%d,x=%d,2:y=%d,x=%d]: r=%d\n",
 1561|      0|                       b->mv[0].y, b->mv[0].x, b->mv[1].y, b->mv[1].x,
 1562|      0|                       ts->msac.rng);
 1563|       |
 1564|       |            // jnt_comp vs. seg vs. wedge
 1565|   154k|            int is_segwedge = 0;
 1566|   154k|            if (f->seq_hdr->masked_compound) {
  ------------------
  |  Branch (1566:17): [True: 142k, False: 12.0k]
  ------------------
 1567|   142k|                const int mask_ctx = get_mask_comp_ctx(t->a, &t->l, by4, bx4);
 1568|       |
 1569|   142k|                is_segwedge = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   142k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1570|   142k|                                  ts->cdf.m.mask_comp[mask_ctx]);
 1571|   142k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   142k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 142k]
  |  |  ------------------
  |  |   35|   142k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   142k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1572|      0|                    printf("Post-segwedge_vs_jntavg[%d,ctx=%d]: r=%d\n",
 1573|      0|                           is_segwedge, mask_ctx, ts->msac.rng);
 1574|   142k|            }
 1575|       |
 1576|   154k|            if (!is_segwedge) {
  ------------------
  |  Branch (1576:17): [True: 103k, False: 50.6k]
  ------------------
 1577|   103k|                if (f->seq_hdr->jnt_comp) {
  ------------------
  |  Branch (1577:21): [True: 75.5k, False: 27.9k]
  ------------------
 1578|  75.5k|                    const int jnt_ctx =
 1579|  75.5k|                        get_jnt_comp_ctx(f->seq_hdr->order_hint_n_bits,
 1580|  75.5k|                                         f->cur.frame_hdr->frame_offset,
 1581|  75.5k|                                         f->refp[b->ref[0]].p.frame_hdr->frame_offset,
 1582|  75.5k|                                         f->refp[b->ref[1]].p.frame_hdr->frame_offset,
 1583|  75.5k|                                         t->a, &t->l, by4, bx4);
 1584|  75.5k|                    b->comp_type = COMP_INTER_WEIGHTED_AVG +
 1585|  75.5k|                                   dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  75.5k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1586|  75.5k|                                       ts->cdf.m.jnt_comp[jnt_ctx]);
 1587|  75.5k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  75.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 75.5k]
  |  |  ------------------
  |  |   35|  75.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  75.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1588|      0|                        printf("Post-jnt_comp[%d,ctx=%d[ac:%d,ar:%d,lc:%d,lr:%d]]: r=%d\n",
 1589|      0|                               b->comp_type == COMP_INTER_AVG,
 1590|      0|                               jnt_ctx, t->a->comp_type[bx4], t->a->ref[0][bx4],
 1591|      0|                               t->l.comp_type[by4], t->l.ref[0][by4],
 1592|      0|                               ts->msac.rng);
 1593|  75.5k|                } else {
 1594|  27.9k|                    b->comp_type = COMP_INTER_AVG;
 1595|  27.9k|                }
 1596|   103k|            } else {
 1597|  50.6k|                if (wedge_allowed_mask & (1 << bs)) {
  ------------------
  |  Branch (1597:21): [True: 40.4k, False: 10.1k]
  ------------------
 1598|  40.4k|                    const int ctx = dav1d_wedge_ctx_lut[bs];
 1599|  40.4k|                    b->comp_type = COMP_INTER_WEDGE -
 1600|  40.4k|                                   dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  40.4k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1601|  40.4k|                                       ts->cdf.m.wedge_comp[ctx]);
 1602|  40.4k|                    if (b->comp_type == COMP_INTER_WEDGE)
  ------------------
  |  Branch (1602:25): [True: 15.2k, False: 25.2k]
  ------------------
 1603|  15.2k|                        b->wedge_idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  15.2k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1604|  40.4k|                                           ts->cdf.m.wedge_idx[ctx], 15);
 1605|  40.4k|                } else {
 1606|  10.1k|                    b->comp_type = COMP_INTER_SEG;
 1607|  10.1k|                }
 1608|  50.6k|                b->mask_sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  50.6k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
 1609|  50.6k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  50.6k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 50.6k]
  |  |  ------------------
  |  |   35|  50.6k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  50.6k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1610|      0|                    printf("Post-seg/wedge[%d,wedge_idx=%d,sign=%d]: r=%d\n",
 1611|      0|                           b->comp_type == COMP_INTER_WEDGE,
 1612|      0|                           b->wedge_idx, b->mask_sign, ts->msac.rng);
 1613|  50.6k|            }
 1614|   890k|        } else {
 1615|   890k|            b->comp_type = COMP_INTER_NONE;
 1616|       |
 1617|       |            // ref
 1618|   890k|            if (seg && seg->ref > 0) {
  ------------------
  |  Branch (1618:17): [True: 463k, False: 426k]
  |  Branch (1618:24): [True: 328k, False: 135k]
  ------------------
 1619|   328k|                b->ref[0] = seg->ref - 1;
 1620|   561k|            } else if (seg && (seg->globalmv || seg->skip)) {
  ------------------
  |  Branch (1620:24): [True: 135k, False: 426k]
  |  Branch (1620:32): [True: 56.4k, False: 78.6k]
  |  Branch (1620:49): [True: 19.8k, False: 58.7k]
  ------------------
 1621|  76.2k|                b->ref[0] = 0;
 1622|   485k|            } else {
 1623|   485k|                const int ctx1 = av1_get_ref_ctx(t->a, &t->l, by4, bx4,
 1624|   485k|                                                 have_top, have_left);
 1625|   485k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   485k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1625:21): [True: 205k, False: 280k]
  ------------------
 1626|   485k|                                                 ts->cdf.m.ref[0][ctx1]))
 1627|   205k|                {
 1628|   205k|                    const int ctx2 = av1_get_ref_2_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  275|   205k|#define av1_get_ref_2_ctx av1_get_bwd_ref_ctx
  ------------------
 1629|   205k|                                                       have_top, have_left);
 1630|   205k|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   205k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1630:25): [True: 153k, False: 51.8k]
  ------------------
 1631|   205k|                                                     ts->cdf.m.ref[1][ctx2]))
 1632|   153k|                    {
 1633|   153k|                        b->ref[0] = 6;
 1634|   153k|                    } else {
 1635|  51.8k|                        const int ctx3 = av1_get_ref_6_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  279|  51.8k|#define av1_get_ref_6_ctx av1_get_bwd_ref_1_ctx
  ------------------
 1636|  51.8k|                                                           have_top, have_left);
 1637|  51.8k|                        b->ref[0] = 4 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  51.8k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1638|  51.8k|                                            ts->cdf.m.ref[5][ctx3]);
 1639|  51.8k|                    }
 1640|   280k|                } else {
 1641|   280k|                    const int ctx2 = av1_get_ref_3_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  276|   280k|#define av1_get_ref_3_ctx av1_get_fwd_ref_ctx
  ------------------
 1642|   280k|                                                       have_top, have_left);
 1643|   280k|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   280k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1643:25): [True: 53.5k, False: 226k]
  ------------------
 1644|   280k|                                                     ts->cdf.m.ref[2][ctx2]))
 1645|  53.5k|                    {
 1646|  53.5k|                        const int ctx3 = av1_get_ref_5_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  278|  53.5k|#define av1_get_ref_5_ctx av1_get_fwd_ref_2_ctx
  ------------------
 1647|  53.5k|                                                           have_top, have_left);
 1648|  53.5k|                        b->ref[0] = 2 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  53.5k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1649|  53.5k|                                            ts->cdf.m.ref[4][ctx3]);
 1650|   226k|                    } else {
 1651|   226k|                        const int ctx3 = av1_get_ref_4_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  277|   226k|#define av1_get_ref_4_ctx av1_get_fwd_ref_1_ctx
  ------------------
 1652|   226k|                                                           have_top, have_left);
 1653|   226k|                        b->ref[0] = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   226k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1654|   226k|                                        ts->cdf.m.ref[3][ctx3]);
 1655|   226k|                    }
 1656|   280k|                }
 1657|   485k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   485k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 485k]
  |  |  ------------------
  |  |   35|   485k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   485k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1658|      0|                    printf("Post-ref[%d]: r=%d\n", b->ref[0], ts->msac.rng);
 1659|   485k|            }
 1660|   890k|            b->ref[1] = -1;
 1661|       |
 1662|   890k|            refmvs_candidate mvstack[8];
 1663|   890k|            int n_mvs, ctx;
 1664|   890k|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1665|   890k|                              (union refmvs_refpair) { .ref = { b->ref[0] + 1, -1 }},
 1666|   890k|                              bs, intra_edge_flags, t->by, t->bx);
 1667|       |
 1668|       |            // mode parsing and mv derivation from ref_mvs
 1669|   890k|            if ((seg && (seg->skip || seg->globalmv)) ||
  ------------------
  |  Branch (1669:18): [True: 463k, False: 426k]
  |  Branch (1669:26): [True: 369k, False: 94.2k]
  |  Branch (1669:39): [True: 24.6k, False: 69.6k]
  ------------------
 1670|   496k|                dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   496k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1670:17): [True: 340k, False: 155k]
  ------------------
 1671|   496k|                                             ts->cdf.m.newmv_mode[ctx & 7]))
 1672|   734k|            {
 1673|   734k|                if ((seg && (seg->skip || seg->globalmv)) ||
  ------------------
  |  Branch (1673:22): [True: 443k, False: 290k]
  |  Branch (1673:30): [True: 369k, False: 74.0k]
  |  Branch (1673:43): [True: 24.6k, False: 49.4k]
  ------------------
 1674|   340k|                    !dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   340k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1674:21): [True: 16.8k, False: 323k]
  ------------------
 1675|   340k|                         ts->cdf.m.globalmv_mode[(ctx >> 3) & 1]))
 1676|   410k|                {
 1677|   410k|                    b->inter_mode = GLOBALMV;
 1678|   410k|                    b->mv[0] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[0]],
 1679|   410k|                                          t->bx, t->by, bw4, bh4, f->frame_hdr);
 1680|   410k|                    has_subpel_filter = imin(bw4, bh4) == 1 ||
  ------------------
  |  Branch (1680:41): [True: 124k, False: 285k]
  ------------------
 1681|   285k|                        f->frame_hdr->gmv[b->ref[0]].type == DAV1D_WM_TYPE_TRANSLATION;
  ------------------
  |  Branch (1681:25): [True: 99.7k, False: 186k]
  ------------------
 1682|   410k|                } else {
 1683|   323k|                    has_subpel_filter = 1;
 1684|   323k|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   323k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1684:25): [True: 168k, False: 155k]
  ------------------
 1685|   323k|                            ts->cdf.m.refmv_mode[(ctx >> 4) & 15]))
 1686|   168k|                    { // NEAREST, NEARER, NEAR or NEARISH
 1687|   168k|                        b->inter_mode = NEARMV;
 1688|   168k|                        b->drl_idx = NEARER_DRL;
 1689|   168k|                        if (n_mvs > 2) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1689:29): [True: 66.9k, False: 101k]
  ------------------
 1690|  66.9k|                            const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1691|  66.9k|                            b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  66.9k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1692|  66.9k|                                              ts->cdf.m.drl_bit[drl_ctx_v2]);
 1693|  66.9k|                            if (b->drl_idx == NEAR_DRL && n_mvs > 3) { // NEAR or NEARISH
  ------------------
  |  Branch (1693:33): [True: 37.8k, False: 29.0k]
  |  Branch (1693:59): [True: 22.5k, False: 15.3k]
  ------------------
 1694|  22.5k|                                const int drl_ctx_v3 =
 1695|  22.5k|                                    get_drl_context(mvstack, 2);
 1696|  22.5k|                                b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  22.5k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1697|  22.5k|                                                  ts->cdf.m.drl_bit[drl_ctx_v3]);
 1698|  22.5k|                            }
 1699|  66.9k|                        }
 1700|   168k|                    } else {
 1701|   155k|                        b->inter_mode = NEARESTMV;
 1702|   155k|                        b->drl_idx = NEAREST_DRL;
 1703|   155k|                    }
 1704|   323k|                    assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1704:21): [True: 323k, False: 0]
  |  Branch (1704:21): [True: 323k, False: 0]
  ------------------
 1705|   323k|                    b->mv[0] = mvstack[b->drl_idx].mv.mv[0];
 1706|   323k|                    if (b->drl_idx < NEAR_DRL)
  ------------------
  |  Branch (1706:25): [True: 285k, False: 37.8k]
  ------------------
 1707|   285k|                        fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1708|   323k|                }
 1709|       |
 1710|   734k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   734k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 734k]
  |  |  ------------------
  |  |   35|   734k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   734k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1711|      0|                    printf("Post-intermode[%d,drl=%d,mv=y:%d,x:%d,n_mvs=%d]: r=%d\n",
 1712|      0|                           b->inter_mode, b->drl_idx, b->mv[0].y, b->mv[0].x, n_mvs,
 1713|      0|                           ts->msac.rng);
 1714|   734k|            } else {
 1715|   155k|                has_subpel_filter = 1;
 1716|   155k|                b->inter_mode = NEWMV;
 1717|   155k|                b->drl_idx = NEAREST_DRL;
 1718|   155k|                if (n_mvs > 1) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1718:21): [True: 120k, False: 35.8k]
  ------------------
 1719|   120k|                    const int drl_ctx_v1 = get_drl_context(mvstack, 0);
 1720|   120k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   120k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1721|   120k|                                      ts->cdf.m.drl_bit[drl_ctx_v1]);
 1722|   120k|                    if (b->drl_idx == NEARER_DRL && n_mvs > 2) { // NEAR or NEARISH
  ------------------
  |  Branch (1722:25): [True: 57.8k, False: 62.2k]
  |  Branch (1722:53): [True: 35.5k, False: 22.3k]
  ------------------
 1723|  35.5k|                        const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1724|  35.5k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  35.5k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1725|  35.5k|                                          ts->cdf.m.drl_bit[drl_ctx_v2]);
 1726|  35.5k|                    }
 1727|   120k|                }
 1728|   155k|                assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1728:17): [True: 155k, False: 0]
  |  Branch (1728:17): [True: 155k, False: 0]
  ------------------
 1729|   155k|                if (n_mvs > 1) {
  ------------------
  |  Branch (1729:21): [True: 120k, False: 35.8k]
  ------------------
 1730|   120k|                    b->mv[0] = mvstack[b->drl_idx].mv.mv[0];
 1731|   120k|                } else {
 1732|  35.8k|                    assert(!b->drl_idx);
  ------------------
  |  Branch (1732:21): [True: 35.8k, False: 0]
  ------------------
 1733|  35.8k|                    b->mv[0] = mvstack[0].mv.mv[0];
 1734|  35.8k|                    fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1735|  35.8k|                }
 1736|   155k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   155k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 155k]
  |  |  ------------------
  |  |   35|   155k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   155k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1737|      0|                    printf("Post-intermode[%d,drl=%d]: r=%d\n",
 1738|      0|                           b->inter_mode, b->drl_idx, ts->msac.rng);
 1739|   155k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv;
 1740|   155k|                read_mv_residual(ts, &b->mv[0], mv_prec);
 1741|   155k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   155k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 155k]
  |  |  ------------------
  |  |   35|   155k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   155k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1742|      0|                    printf("Post-residualmv[mv=y:%d,x:%d]: r=%d\n",
 1743|      0|                           b->mv[0].y, b->mv[0].x, ts->msac.rng);
 1744|   155k|            }
 1745|       |
 1746|       |            // interintra flags
 1747|   890k|            const int ii_sz_grp = dav1d_ymode_size_context[bs];
 1748|   890k|            if (f->seq_hdr->inter_intra &&
  ------------------
  |  Branch (1748:17): [True: 721k, False: 169k]
  ------------------
 1749|   721k|                interintra_allowed_mask & (1 << bs) &&
  ------------------
  |  Branch (1749:17): [True: 314k, False: 406k]
  ------------------
 1750|   314k|                dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   314k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1750:17): [True: 41.5k, False: 272k]
  ------------------
 1751|   314k|                                             ts->cdf.m.interintra[ii_sz_grp]))
 1752|  41.5k|            {
 1753|  41.5k|                b->interintra_mode = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|  41.5k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1754|  41.5k|                                         ts->cdf.m.interintra_mode[ii_sz_grp],
 1755|  41.5k|                                         N_INTER_INTRA_PRED_MODES - 1);
 1756|  41.5k|                const int wedge_ctx = dav1d_wedge_ctx_lut[bs];
 1757|  41.5k|                b->interintra_type = INTER_INTRA_BLEND +
 1758|  41.5k|                                     dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  41.5k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1759|  41.5k|                                         ts->cdf.m.interintra_wedge[wedge_ctx]);
 1760|  41.5k|                if (b->interintra_type == INTER_INTRA_WEDGE)
  ------------------
  |  Branch (1760:21): [True: 9.78k, False: 31.7k]
  ------------------
 1761|  9.78k|                    b->wedge_idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  9.78k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1762|  41.5k|                                       ts->cdf.m.wedge_idx[wedge_ctx], 15);
 1763|   848k|            } else {
 1764|   848k|                b->interintra_type = INTER_INTRA_NONE;
 1765|   848k|            }
 1766|   890k|            if (DEBUG_BLOCK_INFO && f->seq_hdr->inter_intra &&
  ------------------
  |  |   34|   890k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 890k]
  |  |  ------------------
  |  |   35|   890k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   890k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (1766:37): [True: 0, False: 0]
  ------------------
 1767|      0|                interintra_allowed_mask & (1 << bs))
  ------------------
  |  Branch (1767:17): [True: 0, False: 0]
  ------------------
 1768|      0|            {
 1769|      0|                printf("Post-interintra[t=%d,m=%d,w=%d]: r=%d\n",
 1770|      0|                       b->interintra_type, b->interintra_mode,
 1771|      0|                       b->wedge_idx, ts->msac.rng);
 1772|      0|            }
 1773|       |
 1774|       |            // motion variation
 1775|   890k|            if (f->frame_hdr->switchable_motion_mode &&
  ------------------
  |  Branch (1775:17): [True: 793k, False: 96.2k]
  ------------------
 1776|   793k|                b->interintra_type == INTER_INTRA_NONE && imin(bw4, bh4) >= 2 &&
  ------------------
  |  Branch (1776:17): [True: 757k, False: 36.7k]
  |  Branch (1776:59): [True: 476k, False: 281k]
  ------------------
 1777|       |                // is not warped global motion
 1778|   476k|                !(!f->frame_hdr->force_integer_mv && b->inter_mode == GLOBALMV &&
  ------------------
  |  Branch (1778:19): [True: 362k, False: 113k]
  |  Branch (1778:54): [True: 173k, False: 189k]
  ------------------
 1779|   173k|                  f->frame_hdr->gmv[b->ref[0]].type > DAV1D_WM_TYPE_TRANSLATION) &&
  ------------------
  |  Branch (1779:19): [True: 30.7k, False: 142k]
  ------------------
 1780|       |                // has overlappable neighbours
 1781|   445k|                ((have_left && findoddzero(&t->l.intra[by4 + 1], h4 >> 1)) ||
  ------------------
  |  Branch (1781:19): [True: 407k, False: 38.3k]
  |  Branch (1781:32): [True: 396k, False: 10.5k]
  ------------------
 1782|  48.9k|                 (have_top && findoddzero(&t->a->intra[bx4 + 1], w4 >> 1))))
  ------------------
  |  Branch (1782:19): [True: 39.8k, False: 9.10k]
  |  Branch (1782:31): [True: 36.5k, False: 3.27k]
  ------------------
 1783|   432k|            {
 1784|       |                // reaching here means the block allows obmc - check warp by
 1785|       |                // finding matching-ref blocks in top/left edges
 1786|   432k|                uint64_t mask[2] = { 0, 0 };
 1787|   432k|                find_matching_ref(t, intra_edge_flags, bw4, bh4, w4, h4,
 1788|   432k|                                  have_left, have_top, b->ref[0], mask);
 1789|   432k|                const int allow_warp = !f->svc[b->ref[0]][0].scale &&
  ------------------
  |  Branch (1789:40): [True: 301k, False: 131k]
  ------------------
 1790|   301k|                    !f->frame_hdr->force_integer_mv &&
  ------------------
  |  Branch (1790:21): [True: 294k, False: 6.51k]
  ------------------
 1791|   294k|                    f->frame_hdr->warp_motion && (mask[0] | mask[1]);
  ------------------
  |  Branch (1791:21): [True: 233k, False: 61.4k]
  |  Branch (1791:50): [True: 220k, False: 12.6k]
  ------------------
 1792|       |
 1793|   432k|                b->motion_mode = allow_warp ?
  ------------------
  |  Branch (1793:34): [True: 220k, False: 212k]
  ------------------
 1794|   220k|                    dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   220k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1795|   220k|                        ts->cdf.m.motion_mode[bs], 2) :
 1796|   432k|                    dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.obmc[bs]);
  ------------------
  |  |   52|   212k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1797|   432k|                if (b->motion_mode == MM_WARP) {
  ------------------
  |  Branch (1797:21): [True: 79.8k, False: 353k]
  ------------------
 1798|  79.8k|                    has_subpel_filter = 0;
 1799|  79.8k|                    derive_warpmv(t, bw4, bh4, mask, b->mv[0], &t->warpmv);
 1800|  79.8k|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
 1801|  79.8k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  79.8k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 79.8k]
  |  |  ------------------
  |  |   35|  79.8k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  79.8k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1802|      0|                        printf("[ %c%x %c%x %c%x\n  %c%x %c%x %c%x ]\n"
 1803|      0|                               "alpha=%c%x, beta=%c%x, gamma=%c%x, delta=%c%x, "
 1804|      0|                               "mv=y:%d,x:%d\n",
 1805|      0|                               signabs(t->warpmv.matrix[0]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1806|      0|                               signabs(t->warpmv.matrix[1]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1807|      0|                               signabs(t->warpmv.matrix[2]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1808|      0|                               signabs(t->warpmv.matrix[3]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1809|      0|                               signabs(t->warpmv.matrix[4]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1810|      0|                               signabs(t->warpmv.matrix[5]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1811|      0|                               signabs(t->warpmv.u.p.alpha),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1812|      0|                               signabs(t->warpmv.u.p.beta),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1813|      0|                               signabs(t->warpmv.u.p.gamma),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1814|      0|                               signabs(t->warpmv.u.p.delta),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1815|      0|                               b->mv[0].y, b->mv[0].x);
 1816|  79.8k|#undef signabs
 1817|  79.8k|                    if (t->frame_thread.pass) {
  ------------------
  |  Branch (1817:25): [True: 0, False: 79.8k]
  ------------------
 1818|      0|                        if (t->warpmv.type == DAV1D_WM_TYPE_AFFINE) {
  ------------------
  |  Branch (1818:29): [True: 0, False: 0]
  ------------------
 1819|      0|                            b->matrix[0] = t->warpmv.matrix[2] - 0x10000;
 1820|      0|                            b->matrix[1] = t->warpmv.matrix[3];
 1821|      0|                            b->matrix[2] = t->warpmv.matrix[4];
 1822|      0|                            b->matrix[3] = t->warpmv.matrix[5] - 0x10000;
 1823|      0|                        } else {
 1824|      0|                            b->matrix[0] = INT16_MIN;
 1825|      0|                        }
 1826|      0|                    }
 1827|  79.8k|                }
 1828|       |
 1829|   432k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   432k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 432k]
  |  |  ------------------
  |  |   35|   432k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   432k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1830|      0|                    printf("Post-motionmode[%d]: r=%d [mask: 0x%" PRIx64 "/0x%"
 1831|      0|                           PRIx64 "]\n", b->motion_mode, ts->msac.rng, mask[0],
 1832|      0|                            mask[1]);
 1833|   457k|            } else {
 1834|   457k|                b->motion_mode = MM_TRANSLATION;
 1835|   457k|            }
 1836|   890k|        }
 1837|       |
 1838|       |        // subpel filter
 1839|  1.05M|        enum Dav1dFilterMode filter[2];
 1840|  1.05M|        if (f->frame_hdr->subpel_filter_mode == DAV1D_FILTER_SWITCHABLE) {
  ------------------
  |  Branch (1840:13): [True: 489k, False: 569k]
  ------------------
 1841|   489k|            if (has_subpel_filter) {
  ------------------
  |  Branch (1841:17): [True: 320k, False: 169k]
  ------------------
 1842|   320k|                const int comp = b->comp_type != COMP_INTER_NONE;
 1843|   320k|                const int ctx1 = get_filter_ctx(t->a, &t->l, comp, 0, b->ref[0],
 1844|   320k|                                                by4, bx4);
 1845|   320k|                filter[0] = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   320k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1846|   320k|                               ts->cdf.m.filter[0][ctx1],
 1847|   320k|                               DAV1D_N_SWITCHABLE_FILTERS - 1);
 1848|   320k|                if (f->seq_hdr->dual_filter) {
  ------------------
  |  Branch (1848:21): [True: 261k, False: 58.9k]
  ------------------
 1849|   261k|                    const int ctx2 = get_filter_ctx(t->a, &t->l, comp, 1,
 1850|   261k|                                                    b->ref[0], by4, bx4);
 1851|   261k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   261k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 261k]
  |  |  ------------------
  |  |   35|   261k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   261k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1852|      0|                        printf("Post-subpel_filter1[%d,ctx=%d]: r=%d\n",
 1853|      0|                               filter[0], ctx1, ts->msac.rng);
 1854|   261k|                    filter[1] = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   261k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1855|   261k|                                    ts->cdf.m.filter[1][ctx2],
 1856|   261k|                                    DAV1D_N_SWITCHABLE_FILTERS - 1);
 1857|   261k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   261k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 261k]
  |  |  ------------------
  |  |   35|   261k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   261k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1858|      0|                        printf("Post-subpel_filter2[%d,ctx=%d]: r=%d\n",
 1859|      0|                               filter[1], ctx2, ts->msac.rng);
 1860|   261k|                } else {
 1861|  58.9k|                    filter[1] = filter[0];
 1862|  58.9k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  58.9k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 58.9k]
  |  |  ------------------
  |  |   35|  58.9k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  58.9k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1863|      0|                        printf("Post-subpel_filter[%d,ctx=%d]: r=%d\n",
 1864|      0|                               filter[0], ctx1, ts->msac.rng);
 1865|  58.9k|                }
 1866|   320k|            } else {
 1867|   169k|                filter[0] = filter[1] = DAV1D_FILTER_8TAP_REGULAR;
 1868|   169k|            }
 1869|   569k|        } else {
 1870|   569k|            filter[0] = filter[1] = f->frame_hdr->subpel_filter_mode;
 1871|   569k|        }
 1872|  1.05M|        b->filter2d = dav1d_filter_2d[filter[1]][filter[0]];
 1873|       |
 1874|  1.05M|        read_vartx_tree(t, b, bs, bx4, by4);
 1875|       |
 1876|       |        // reconstruction
 1877|  1.05M|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1877:13): [True: 0, False: 1.05M]
  ------------------
 1878|      0|            f->bd_fn.read_coef_blocks(t, bs, b);
 1879|  1.05M|        } else {
 1880|  1.05M|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (1880:17): [True: 0, False: 1.05M]
  ------------------
 1881|  1.05M|        }
 1882|       |
 1883|  1.05M|        if (f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (1883:13): [True: 677k, False: 381k]
  ------------------
 1884|   381k|            f->frame_hdr->loopfilter.level_y[1])
  ------------------
  |  Branch (1884:13): [True: 97.0k, False: 284k]
  ------------------
 1885|   774k|        {
 1886|   774k|            const int is_globalmv =
 1887|   774k|                b->inter_mode == (is_comp ? GLOBALMV_GLOBALMV : GLOBALMV);
  ------------------
  |  Branch (1887:35): [True: 79.9k, False: 694k]
  ------------------
 1888|   774k|            const uint8_t (*const lf_lvls)[8][2] = (const uint8_t (*)[8][2])
 1889|   774k|                &ts->lflvl[b->seg_id][0][b->ref[0] + 1][!is_globalmv];
 1890|   774k|            const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1891|   774k|            enum RectTxfmSize ytx = b->max_ytx, uvtx = b->uvtx;
 1892|   774k|            if (f->frame_hdr->segmentation.lossless[b->seg_id]) {
  ------------------
  |  Branch (1892:17): [True: 1.80k, False: 772k]
  ------------------
 1893|  1.80k|                ytx  = (enum RectTxfmSize) TX_4X4;
 1894|  1.80k|                uvtx = (enum RectTxfmSize) TX_4X4;
 1895|  1.80k|            }
 1896|   774k|            dav1d_create_lf_mask_inter(t->lf_mask, f->lf.level, f->b4_stride, lf_lvls,
 1897|   774k|                                       t->bx, t->by, f->w4, f->h4, b->skip, bs,
 1898|   774k|                                       ytx, tx_split, uvtx, f->cur.p.layout,
 1899|   774k|                                       &t->a->tx_lpf_y[bx4], &t->l.tx_lpf_y[by4],
 1900|   774k|                                       has_chroma ? &t->a->tx_lpf_uv[cbx4] : NULL,
  ------------------
  |  Branch (1900:40): [True: 253k, False: 521k]
  ------------------
 1901|   774k|                                       has_chroma ? &t->l.tx_lpf_uv[cby4] : NULL);
  ------------------
  |  Branch (1901:40): [True: 253k, False: 521k]
  ------------------
 1902|   774k|        }
 1903|       |
 1904|       |        // context updates
 1905|  1.05M|        if (is_comp)
  ------------------
  |  Branch (1905:13): [True: 168k, False: 890k]
  ------------------
 1906|   168k|            splat_tworef_mv(f->c, t, bs, b, bw4, bh4);
 1907|   890k|        else
 1908|   890k|            splat_oneref_mv(f->c, t, bs, b, bw4, bh4);
 1909|  1.05M|        BlockContext *edge = t->a;
 1910|  3.17M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1910:36): [True: 2.11M, False: 1.05M]
  ------------------
 1911|  2.11M|#define set_ctx(rep_macro) \
 1912|  2.11M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1913|  2.11M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
 1914|  2.11M|            rep_macro(edge->intra, off, 0); \
 1915|  2.11M|            rep_macro(edge->skip, off, b->skip); \
 1916|  2.11M|            rep_macro(edge->pal_sz, off, 0); \
 1917|       |            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
 1918|  2.11M|            rep_macro(t->pal_sz_uv[i], off, 0); \
 1919|  2.11M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
 1920|  2.11M|            rep_macro(edge->comp_type, off, b->comp_type); \
 1921|  2.11M|            rep_macro(edge->filter[0], off, filter[0]); \
 1922|  2.11M|            rep_macro(edge->filter[1], off, filter[1]); \
 1923|  2.11M|            rep_macro(edge->mode, off, b->inter_mode); \
 1924|  2.11M|            rep_macro(edge->ref[0], off, b->ref[0]); \
 1925|  2.11M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
 1926|  2.11M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  2.11M|    switch (var) { \
  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1912|   409k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   409k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   409k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   409k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   409k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   409k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   409k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   409k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   409k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   409k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   409k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   409k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   409k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   409k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   409k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   409k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 409k, False: 1.70M]
  |  |  ------------------
  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1912|   672k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   672k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   672k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   672k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   672k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   672k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   672k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   672k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   672k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   672k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   672k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   672k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   672k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   672k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   672k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   672k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 672k, False: 1.44M]
  |  |  ------------------
  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1912|   535k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   535k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   535k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   535k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   535k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   535k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   535k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   535k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   535k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   535k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   535k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   535k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   535k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   535k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   535k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   535k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 535k, False: 1.58M]
  |  |  ------------------
  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1912|   229k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   229k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   229k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   229k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   229k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   229k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   229k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   229k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   229k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   229k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   229k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   229k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   229k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   229k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   229k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   229k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 229k, False: 1.88M]
  |  |  ------------------
  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1912|   191k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   191k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   191k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   191k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   191k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   191k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   191k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   191k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   191k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   191k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   191k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   191k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   191k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   191k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   191k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   191k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   191k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   191k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 191k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 191k, False: 1.92M]
  |  |  ------------------
  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1912|  79.2k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|  79.2k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|  79.2k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|  79.2k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|  79.2k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|  79.2k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|  79.2k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|  79.2k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|  79.2k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|  79.2k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|  79.2k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|  79.2k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|  79.2k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|  79.2k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  79.2k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  79.2k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  79.2k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  79.2k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 79.2k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 79.2k, False: 2.03M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 2.11M]
  |  |  ------------------
  |  |   78|  2.11M|    }
  ------------------
  |  Branch (1926:13): [Folded, False: 0]
  ------------------
 1927|  2.11M|#undef set_ctx
 1928|  2.11M|        }
 1929|  1.05M|        if (has_chroma) {
  ------------------
  |  Branch (1929:13): [True: 484k, False: 574k]
  ------------------
 1930|   484k|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
 1931|   484k|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
 1932|   484k|        }
 1933|  1.05M|    }
 1934|       |
 1935|       |    // update contexts
 1936|  4.10M|    if (f->frame_hdr->segmentation.enabled &&
  ------------------
  |  Branch (1936:9): [True: 1.39M, False: 2.70M]
  ------------------
 1937|  1.39M|        f->frame_hdr->segmentation.update_map)
  ------------------
  |  Branch (1937:9): [True: 1.07M, False: 313k]
  ------------------
 1938|  1.07M|    {
 1939|  1.07M|        uint8_t *seg_ptr = &f->cur_segmap[t->by * f->b4_stride + t->bx];
 1940|  1.07M|#define set_ctx(rep_macro) \
 1941|  1.07M|        for (int y = 0; y < bh4; y++) { \
 1942|  1.07M|            rep_macro(seg_ptr, 0, b->seg_id); \
 1943|  1.07M|            seg_ptr += f->b4_stride; \
 1944|  1.07M|        }
 1945|  1.07M|        case_set(b_dim[2]);
  ------------------
  |  |   70|  1.07M|    switch (var) { \
  |  |   71|   346k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1941|   816k|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 470k, False: 346k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   470k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   470k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   470k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   470k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   470k|        }
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 346k, False: 733k]
  |  |  ------------------
  |  |   72|   190k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1941|   764k|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 574k, False: 190k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   574k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   574k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   574k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   574k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   574k|        }
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 190k, False: 889k]
  |  |  ------------------
  |  |   73|   228k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1941|  1.03M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 804k, False: 228k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   804k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   804k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   804k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   804k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   804k|        }
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 228k, False: 851k]
  |  |  ------------------
  |  |   74|   100k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1941|   672k|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 572k, False: 100k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   572k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   572k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   572k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   572k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   572k|        }
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 100k, False: 979k]
  |  |  ------------------
  |  |   75|   125k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1941|  2.03M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 1.90M, False: 125k]
  |  |  |  |  ------------------
  |  |  |  | 1942|  1.90M|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.90M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.90M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.90M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.90M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.90M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|  1.90M|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|  1.90M|        }
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 125k, False: 954k]
  |  |  ------------------
  |  |   76|  89.0k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1941|  2.90M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 2.81M, False: 89.0k]
  |  |  |  |  ------------------
  |  |  |  | 1942|  2.81M|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  2.81M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  2.81M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  2.81M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  2.81M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 2.81M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|  2.81M|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|  2.81M|        }
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 89.0k, False: 990k]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 1.07M]
  |  |  ------------------
  |  |   78|  1.07M|    }
  ------------------
  |  Branch (1945:9): [Folded, False: 0]
  ------------------
 1946|  1.07M|#undef set_ctx
 1947|  1.07M|    }
 1948|  4.10M|    if (!b->skip) {
  ------------------
  |  Branch (1948:9): [True: 2.20M, False: 1.89M]
  ------------------
 1949|  2.20M|        uint16_t (*noskip_mask)[2] = &t->lf_mask->noskip_mask[by4 >> 1];
 1950|  2.20M|        const unsigned mask = (~0U >> (32 - bw4)) << (bx4 & 15);
 1951|  2.20M|        const int bx_idx = (bx4 & 16) >> 4;
 1952|  6.88M|        for (int y = 0; y < bh4; y += 2, noskip_mask++) {
  ------------------
  |  Branch (1952:25): [True: 4.67M, False: 2.20M]
  ------------------
 1953|  4.67M|            (*noskip_mask)[bx_idx] |= mask;
 1954|  4.67M|            if (bw4 == 32) // this should be mask >> 16, but it's 0xffffffff anyway
  ------------------
  |  Branch (1954:17): [True: 394k, False: 4.28M]
  ------------------
 1955|   394k|                (*noskip_mask)[1] |= mask;
 1956|  4.67M|        }
 1957|  2.20M|    }
 1958|       |
 1959|  4.10M|    if (t->frame_thread.pass == 1 && !b->intra && IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|      0|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (1959:9): [True: 0, False: 4.10M]
  |  Branch (1959:38): [True: 0, False: 0]
  ------------------
 1960|      0|        const int sby = (t->by - ts->tiling.row_start) >> f->sb_shift;
 1961|      0|        int (*const lowest_px)[2] = ts->lowest_pixel[sby];
 1962|       |
 1963|       |        // keep track of motion vectors for each reference
 1964|      0|        if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1964:13): [True: 0, False: 0]
  ------------------
 1965|       |            // y
 1966|      0|            if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1966:17): [True: 0, False: 0]
  ------------------
 1967|      0|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1967:19): [True: 0, False: 0]
  |  Branch (1967:48): [True: 0, False: 0]
  ------------------
 1968|      0|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1968:19): [True: 0, False: 0]
  |  Branch (1968:48): [True: 0, False: 0]
  ------------------
 1969|      0|            {
 1970|      0|                affine_lowest_px_luma(t, &lowest_px[b->ref[0]][0], b_dim,
 1971|      0|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1971:39): [True: 0, False: 0]
  ------------------
 1972|      0|                                      &f->frame_hdr->gmv[b->ref[0]]);
 1973|      0|            } else {
 1974|      0|                mc_lowest_px(&lowest_px[b->ref[0]][0], t->by, bh4, b->mv[0].y,
 1975|      0|                             0, &f->svc[b->ref[0]][1]);
 1976|      0|                if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1976:21): [True: 0, False: 0]
  ------------------
 1977|      0|                    obmc_lowest_px(t, lowest_px, 0, b_dim, bx4, by4, w4, h4);
 1978|      0|                }
 1979|      0|            }
 1980|       |
 1981|       |            // uv
 1982|      0|            if (has_chroma) {
  ------------------
  |  Branch (1982:17): [True: 0, False: 0]
  ------------------
 1983|       |                // sub8x8 derivation
 1984|      0|                int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1984:33): [True: 0, False: 0]
  |  Branch (1984:50): [True: 0, False: 0]
  ------------------
 1985|      0|                refmvs_block *const *r;
 1986|      0|                if (is_sub8x8) {
  ------------------
  |  Branch (1986:21): [True: 0, False: 0]
  ------------------
 1987|      0|                    assert(ss_hor == 1);
  ------------------
  |  Branch (1987:21): [True: 0, False: 0]
  ------------------
 1988|      0|                    r = &t->rt.r[(t->by & 31) + 5];
 1989|      0|                    if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1989:25): [True: 0, False: 0]
  ------------------
 1990|      0|                    if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1990:25): [True: 0, False: 0]
  ------------------
 1991|      0|                    if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1991:25): [True: 0, False: 0]
  |  Branch (1991:37): [True: 0, False: 0]
  ------------------
 1992|      0|                        is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1993|      0|                }
 1994|       |
 1995|       |                // chroma prediction
 1996|      0|                if (is_sub8x8) {
  ------------------
  |  Branch (1996:21): [True: 0, False: 0]
  ------------------
 1997|      0|                    assert(ss_hor == 1);
  ------------------
  |  Branch (1997:21): [True: 0, False: 0]
  ------------------
 1998|      0|                    if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1998:25): [True: 0, False: 0]
  |  Branch (1998:37): [True: 0, False: 0]
  ------------------
 1999|      0|                        const refmvs_block *const rr = &r[-1][t->bx - 1];
 2000|      0|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2001|      0|                                     t->by - 1, bh4, rr->mv.mv[0].y, ss_ver,
 2002|      0|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2003|      0|                    }
 2004|      0|                    if (bw4 == 1) {
  ------------------
  |  Branch (2004:25): [True: 0, False: 0]
  ------------------
 2005|      0|                        const refmvs_block *const rr = &r[0][t->bx - 1];
 2006|      0|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2007|      0|                                     t->by, bh4, rr->mv.mv[0].y, ss_ver,
 2008|      0|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2009|      0|                    }
 2010|      0|                    if (bh4 == ss_ver) {
  ------------------
  |  Branch (2010:25): [True: 0, False: 0]
  ------------------
 2011|      0|                        const refmvs_block *const rr = &r[-1][t->bx];
 2012|      0|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2013|      0|                                     t->by - 1, bh4, rr->mv.mv[0].y, ss_ver,
 2014|      0|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2015|      0|                    }
 2016|      0|                    mc_lowest_px(&lowest_px[b->ref[0]][1], t->by, bh4,
 2017|      0|                                 b->mv[0].y, ss_ver, &f->svc[b->ref[0]][1]);
 2018|      0|                } else {
 2019|      0|                    if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (2019:25): [True: 0, False: 0]
  ------------------
 2020|      0|                        ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (2020:27): [True: 0, False: 0]
  |  Branch (2020:56): [True: 0, False: 0]
  ------------------
 2021|      0|                         (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (2021:27): [True: 0, False: 0]
  |  Branch (2021:56): [True: 0, False: 0]
  ------------------
 2022|      0|                    {
 2023|      0|                        affine_lowest_px_chroma(t, &lowest_px[b->ref[0]][1], b_dim,
 2024|      0|                                                b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (2024:49): [True: 0, False: 0]
  ------------------
 2025|      0|                                                &f->frame_hdr->gmv[b->ref[0]]);
 2026|      0|                    } else {
 2027|      0|                        mc_lowest_px(&lowest_px[b->ref[0]][1],
 2028|      0|                                     t->by & ~ss_ver, bh4 << (bh4 == ss_ver),
 2029|      0|                                     b->mv[0].y, ss_ver, &f->svc[b->ref[0]][1]);
 2030|      0|                        if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (2030:29): [True: 0, False: 0]
  ------------------
 2031|      0|                            obmc_lowest_px(t, lowest_px, 1, b_dim, bx4, by4, w4, h4);
 2032|      0|                        }
 2033|      0|                    }
 2034|      0|                }
 2035|      0|            }
 2036|      0|        } else {
 2037|       |            // y
 2038|      0|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (2038:29): [True: 0, False: 0]
  ------------------
 2039|      0|                if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (2039:21): [True: 0, False: 0]
  |  Branch (2039:59): [True: 0, False: 0]
  ------------------
 2040|      0|                    affine_lowest_px_luma(t, &lowest_px[b->ref[i]][0], b_dim,
 2041|      0|                                          &f->frame_hdr->gmv[b->ref[i]]);
 2042|      0|                } else {
 2043|      0|                    mc_lowest_px(&lowest_px[b->ref[i]][0], t->by, bh4,
 2044|      0|                                 b->mv[i].y, 0, &f->svc[b->ref[i]][1]);
 2045|      0|                }
 2046|      0|            }
 2047|       |
 2048|       |            // uv
 2049|      0|            if (has_chroma) for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (2049:17): [True: 0, False: 0]
  |  Branch (2049:45): [True: 0, False: 0]
  ------------------
 2050|      0|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (2050:21): [True: 0, False: 0]
  ------------------
 2051|      0|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (2051:21): [True: 0, False: 0]
  |  Branch (2051:45): [True: 0, False: 0]
  ------------------
 2052|      0|                {
 2053|      0|                    affine_lowest_px_chroma(t, &lowest_px[b->ref[i]][1], b_dim,
 2054|      0|                                            &f->frame_hdr->gmv[b->ref[i]]);
 2055|      0|                } else {
 2056|      0|                    mc_lowest_px(&lowest_px[b->ref[i]][1], t->by, bh4,
 2057|      0|                                 b->mv[i].y, ss_ver, &f->svc[b->ref[i]][1]);
 2058|      0|                }
 2059|      0|            }
 2060|      0|        }
 2061|      0|    }
 2062|       |
 2063|  4.10M|    return 0;
 2064|  4.10M|}
decode.c:get_prev_frame_segid:
  499|   219k|{
  500|   219k|    assert(f->frame_hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (500:5): [True: 219k, False: 0]
  ------------------
  501|       |
  502|   219k|    unsigned seg_id = 8;
  503|   219k|    ref_seg_map += by * stride + bx;
  504|   230k|    do {
  505|  1.46M|        for (int x = 0; x < w4; x++)
  ------------------
  |  Branch (505:25): [True: 1.23M, False: 230k]
  ------------------
  506|  1.23M|            seg_id = imin(seg_id, ref_seg_map[x]);
  507|   230k|        ref_seg_map += stride;
  508|   230k|    } while (--h4 > 0 && seg_id);
  ------------------
  |  Branch (508:14): [True: 179k, False: 50.5k]
  |  Branch (508:26): [True: 11.1k, False: 168k]
  ------------------
  509|   219k|    assert(seg_id < 8);
  ------------------
  |  Branch (509:5): [True: 219k, False: 0]
  ------------------
  510|       |
  511|   219k|    return seg_id;
  512|   219k|}
decode.c:neg_deinterleave:
  169|   700k|static int neg_deinterleave(int diff, int ref, int max) {
  170|   700k|    if (!ref) return diff;
  ------------------
  |  Branch (170:9): [True: 384k, False: 316k]
  ------------------
  171|   316k|    if (ref >= (max - 1)) return max - diff - 1;
  ------------------
  |  Branch (171:9): [True: 69.1k, False: 247k]
  ------------------
  172|   247k|    if (2 * ref < max) {
  ------------------
  |  Branch (172:9): [True: 158k, False: 88.4k]
  ------------------
  173|   158k|        if (diff <= 2 * ref) {
  ------------------
  |  Branch (173:13): [True: 126k, False: 32.6k]
  ------------------
  174|   126k|            if (diff & 1)
  ------------------
  |  Branch (174:17): [True: 15.2k, False: 110k]
  ------------------
  175|  15.2k|                return ref + ((diff + 1) >> 1);
  176|   110k|            else
  177|   110k|                return ref - (diff >> 1);
  178|   126k|        }
  179|  32.6k|        return diff;
  180|   158k|    } else {
  181|  88.4k|        if (diff <= 2 * (max - ref - 1)) {
  ------------------
  |  Branch (181:13): [True: 73.0k, False: 15.3k]
  ------------------
  182|  73.0k|            if (diff & 1)
  ------------------
  |  Branch (182:17): [True: 11.9k, False: 61.1k]
  ------------------
  183|  11.9k|                return ref + ((diff + 1) >> 1);
  184|  61.1k|            else
  185|  61.1k|                return ref - (diff >> 1);
  186|  73.0k|        }
  187|  15.3k|        return max - (diff + 1);
  188|  88.4k|    }
  189|   247k|}
decode.c:read_pal_indices:
  419|  96.7k|{
  420|  96.7k|    Dav1dTileState *const ts = t->ts;
  421|  96.7k|    const ptrdiff_t stride = bw4 * 4;
  422|  96.7k|    assert(pal_idx);
  ------------------
  |  Branch (422:5): [True: 96.7k, False: 0]
  ------------------
  423|  96.7k|    uint8_t *const pal_tmp = t->scratch.pal_idx_uv;
  424|  96.7k|    pal_tmp[0] = dav1d_msac_decode_uniform(&ts->msac, pal_sz);
  425|  96.7k|    uint16_t (*const color_map_cdf)[8] =
  426|  96.7k|        ts->cdf.m.color_map[pl][pal_sz - 2];
  427|  96.7k|    uint8_t (*const order)[8] = t->scratch.pal_order;
  428|  96.7k|    uint8_t *const ctx = t->scratch.pal_ctx;
  429|  2.52M|    for (int i = 1; i < 4 * (w4 + h4) - 1; i++) {
  ------------------
  |  Branch (429:21): [True: 2.43M, False: 96.7k]
  ------------------
  430|       |        // top/left-to-bottom/right diagonals ("wave-front")
  431|  2.43M|        const int first = imin(i, w4 * 4 - 1);
  432|  2.43M|        const int last = imax(0, i - h4 * 4 + 1);
  433|  2.43M|        order_palette(pal_tmp, stride, i, first, last, order, ctx);
  434|  20.3M|        for (int j = first, m = 0; j >= last; j--, m++) {
  ------------------
  |  Branch (434:36): [True: 17.9M, False: 2.43M]
  ------------------
  435|  17.9M|            const int color_idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  17.9M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  436|  17.9M|                                      color_map_cdf[ctx[m]], pal_sz - 1);
  437|  17.9M|            pal_tmp[(i - j) * stride + j] = order[m][color_idx];
  438|  17.9M|        }
  439|  2.43M|    }
  440|       |
  441|  96.7k|    t->c->pal_dsp.pal_idx_finish(pal_idx, pal_tmp, bw4 * 4, bh4 * 4,
  442|  96.7k|                                 w4 * 4, h4 * 4);
  443|  96.7k|}
decode.c:order_palette:
  356|  2.43M|{
  357|  2.43M|    int have_top = i > first;
  358|       |
  359|  2.43M|    assert(pal_idx);
  ------------------
  |  Branch (359:5): [True: 2.43M, False: 0]
  ------------------
  360|  2.43M|    pal_idx += first + (i - first) * stride;
  361|  20.3M|    for (int j = first, n = 0; j >= last; have_top = 1, j--, n++, pal_idx += stride - 1) {
  ------------------
  |  Branch (361:32): [True: 17.9M, False: 2.43M]
  ------------------
  362|  17.9M|        const int have_left = j > 0;
  363|       |
  364|  17.9M|        assert(have_left || have_top);
  ------------------
  |  Branch (364:9): [True: 16.9M, False: 953k]
  |  Branch (364:9): [True: 953k, False: 0]
  ------------------
  365|       |
  366|  17.9M|#define add(v_in) do { \
  367|  17.9M|        const int v = v_in; \
  368|  17.9M|        assert((unsigned)v < 8U); \
  369|  17.9M|        order[n][o_idx++] = v; \
  370|  17.9M|        mask |= 1 << v; \
  371|  17.9M|    } while (0)
  372|       |
  373|  17.9M|        unsigned mask = 0;
  374|  17.9M|        int o_idx = 0;
  375|  17.9M|        if (!have_left) {
  ------------------
  |  Branch (375:13): [True: 953k, False: 16.9M]
  ------------------
  376|   953k|            ctx[n] = 0;
  377|   953k|            add(pal_idx[-stride]);
  ------------------
  |  |  366|   953k|#define add(v_in) do { \
  |  |  367|   953k|        const int v = v_in; \
  |  |  368|   953k|        assert((unsigned)v < 8U); \
  |  |  369|   953k|        order[n][o_idx++] = v; \
  |  |  370|   953k|        mask |= 1 << v; \
  |  |  371|   953k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 953k]
  |  |  ------------------
  ------------------
  |  Branch (377:13): [True: 953k, False: 0]
  ------------------
  378|  16.9M|        } else if (!have_top) {
  ------------------
  |  Branch (378:20): [True: 1.47M, False: 15.4M]
  ------------------
  379|  1.47M|            ctx[n] = 0;
  380|  1.47M|            add(pal_idx[-1]);
  ------------------
  |  |  366|  1.47M|#define add(v_in) do { \
  |  |  367|  1.47M|        const int v = v_in; \
  |  |  368|  1.47M|        assert((unsigned)v < 8U); \
  |  |  369|  1.47M|        order[n][o_idx++] = v; \
  |  |  370|  1.47M|        mask |= 1 << v; \
  |  |  371|  1.47M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 1.47M]
  |  |  ------------------
  ------------------
  |  Branch (380:13): [True: 1.47M, False: 0]
  ------------------
  381|  15.4M|        } else {
  382|  15.4M|            const int l = pal_idx[-1], t = pal_idx[-stride], tl = pal_idx[-(stride + 1)];
  383|  15.4M|            const int same_t_l = t == l;
  384|  15.4M|            const int same_t_tl = t == tl;
  385|  15.4M|            const int same_l_tl = l == tl;
  386|  15.4M|            const int same_all = same_t_l & same_t_tl & same_l_tl;
  387|       |
  388|  15.4M|            if (same_all) {
  ------------------
  |  Branch (388:17): [True: 8.63M, False: 6.83M]
  ------------------
  389|  8.63M|                ctx[n] = 4;
  390|  8.63M|                add(t);
  ------------------
  |  |  366|  8.63M|#define add(v_in) do { \
  |  |  367|  8.63M|        const int v = v_in; \
  |  |  368|  8.63M|        assert((unsigned)v < 8U); \
  |  |  369|  8.63M|        order[n][o_idx++] = v; \
  |  |  370|  8.63M|        mask |= 1 << v; \
  |  |  371|  8.63M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 8.63M]
  |  |  ------------------
  ------------------
  |  Branch (390:17): [True: 8.63M, False: 0]
  ------------------
  391|  8.63M|            } else if (same_t_l) {
  ------------------
  |  Branch (391:24): [True: 469k, False: 6.36M]
  ------------------
  392|   469k|                ctx[n] = 3;
  393|   469k|                add(t);
  ------------------
  |  |  366|   469k|#define add(v_in) do { \
  |  |  367|   469k|        const int v = v_in; \
  |  |  368|   469k|        assert((unsigned)v < 8U); \
  |  |  369|   469k|        order[n][o_idx++] = v; \
  |  |  370|   469k|        mask |= 1 << v; \
  |  |  371|   469k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 469k]
  |  |  ------------------
  ------------------
  |  Branch (393:17): [True: 469k, False: 0]
  ------------------
  394|   469k|                add(tl);
  ------------------
  |  |  366|   469k|#define add(v_in) do { \
  |  |  367|   469k|        const int v = v_in; \
  |  |  368|   469k|        assert((unsigned)v < 8U); \
  |  |  369|   469k|        order[n][o_idx++] = v; \
  |  |  370|   469k|        mask |= 1 << v; \
  |  |  371|   469k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 469k]
  |  |  ------------------
  ------------------
  |  Branch (394:17): [True: 469k, False: 0]
  ------------------
  395|  6.36M|            } else if (same_t_tl | same_l_tl) {
  ------------------
  |  Branch (395:24): [True: 5.17M, False: 1.18M]
  ------------------
  396|  5.17M|                ctx[n] = 2;
  397|  5.17M|                add(tl);
  ------------------
  |  |  366|  5.17M|#define add(v_in) do { \
  |  |  367|  5.17M|        const int v = v_in; \
  |  |  368|  5.17M|        assert((unsigned)v < 8U); \
  |  |  369|  5.17M|        order[n][o_idx++] = v; \
  |  |  370|  5.17M|        mask |= 1 << v; \
  |  |  371|  5.17M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 5.17M]
  |  |  ------------------
  ------------------
  |  Branch (397:17): [True: 5.17M, False: 0]
  ------------------
  398|  5.17M|                add(same_t_tl ? l : t);
  ------------------
  |  |  366|  5.17M|#define add(v_in) do { \
  |  |  367|  10.3M|        const int v = v_in; \
  |  |  ------------------
  |  |  |  Branch (367:23): [True: 2.66M, False: 2.51M]
  |  |  ------------------
  |  |  368|  5.17M|        assert((unsigned)v < 8U); \
  |  |  369|  5.17M|        order[n][o_idx++] = v; \
  |  |  370|  5.17M|        mask |= 1 << v; \
  |  |  371|  5.17M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 5.17M]
  |  |  ------------------
  ------------------
  |  Branch (398:17): [True: 5.17M, False: 0]
  ------------------
  399|  5.17M|            } else {
  400|  1.18M|                ctx[n] = 1;
  401|  1.18M|                add(imin(t, l));
  ------------------
  |  |  366|  1.18M|#define add(v_in) do { \
  |  |  367|  1.18M|        const int v = v_in; \
  |  |  368|  1.18M|        assert((unsigned)v < 8U); \
  |  |  369|  1.18M|        order[n][o_idx++] = v; \
  |  |  370|  1.18M|        mask |= 1 << v; \
  |  |  371|  1.18M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 1.18M]
  |  |  ------------------
  ------------------
  |  Branch (401:17): [True: 1.18M, False: 0]
  ------------------
  402|  1.18M|                add(imax(t, l));
  ------------------
  |  |  366|  1.18M|#define add(v_in) do { \
  |  |  367|  1.18M|        const int v = v_in; \
  |  |  368|  1.18M|        assert((unsigned)v < 8U); \
  |  |  369|  1.18M|        order[n][o_idx++] = v; \
  |  |  370|  1.18M|        mask |= 1 << v; \
  |  |  371|  1.18M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 1.18M]
  |  |  ------------------
  ------------------
  |  Branch (402:17): [True: 1.18M, False: 0]
  ------------------
  403|  1.18M|                add(tl);
  ------------------
  |  |  366|  1.18M|#define add(v_in) do { \
  |  |  367|  1.18M|        const int v = v_in; \
  |  |  368|  1.18M|        assert((unsigned)v < 8U); \
  |  |  369|  1.18M|        order[n][o_idx++] = v; \
  |  |  370|  1.18M|        mask |= 1 << v; \
  |  |  371|  1.18M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 1.18M]
  |  |  ------------------
  ------------------
  |  Branch (403:17): [True: 1.18M, False: 0]
  ------------------
  404|  1.18M|            }
  405|  15.4M|        }
  406|   161M|        for (unsigned m = 1, bit = 0; m < 0x100; m <<= 1, bit++)
  ------------------
  |  Branch (406:39): [True: 143M, False: 17.9M]
  ------------------
  407|   143M|            if (!(mask & m))
  ------------------
  |  Branch (407:17): [True: 117M, False: 25.9M]
  ------------------
  408|   117M|                order[n][o_idx++] = bit;
  409|       |        assert(o_idx == 8);
  ------------------
  |  Branch (409:9): [True: 17.9M, False: 0]
  ------------------
  410|  17.9M|#undef add
  411|  17.9M|    }
  412|  2.43M|}
decode.c:splat_intraref:
  566|  1.72M|{
  567|  1.72M|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  568|  1.72M|        .ref.ref = { 0, -1 },
  569|  1.72M|        .mv.mv[0].n = INVALID_MV,
  ------------------
  |  |   40|  1.72M|#define INVALID_MV 0x80008000
  ------------------
  570|  1.72M|        .bs = bs,
  571|  1.72M|        .mf = 0,
  572|  1.72M|    };
  573|  1.72M|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  574|  1.72M|}
decode.c:read_mv_residual:
  109|   931k|{
  110|   931k|    MsacContext *const msac = &ts->msac;
  111|   931k|    const enum MVJoint mv_joint =
  112|   931k|        dav1d_msac_decode_symbol_adapt4(msac, ts->cdf.mv.joint, N_MV_JOINTS - 1);
  ------------------
  |  |   47|   931k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  113|   931k|    if (mv_joint & MV_JOINT_V)
  ------------------
  |  Branch (113:9): [True: 807k, False: 124k]
  ------------------
  114|   807k|        ref_mv->y += read_mv_component_diff(msac, &ts->cdf.mv.comp[0], mv_prec);
  115|   931k|    if (mv_joint & MV_JOINT_H)
  ------------------
  |  Branch (115:9): [True: 803k, False: 128k]
  ------------------
  116|   803k|        ref_mv->x += read_mv_component_diff(msac, &ts->cdf.mv.comp[1], mv_prec);
  117|   931k|}
decode.c:read_mv_component_diff:
   79|  1.61M|{
   80|  1.61M|    const int sign = dav1d_msac_decode_bool_adapt(msac, mv_comp->sign);
  ------------------
  |  |   52|  1.61M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   81|  1.61M|    const int cl = dav1d_msac_decode_symbol_adapt16(msac, mv_comp->classes, 10);
  ------------------
  |  |   57|  1.61M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
   82|  1.61M|    int up, fp = 3, hp = 1;
   83|       |
   84|  1.61M|    if (!cl) {
  ------------------
  |  Branch (84:9): [True: 326k, False: 1.28M]
  ------------------
   85|   326k|        up = dav1d_msac_decode_bool_adapt(msac, mv_comp->class0);
  ------------------
  |  |   52|   326k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   86|   326k|        if (mv_prec >= 0) {  // !force_integer_mv
  ------------------
  |  Branch (86:13): [True: 165k, False: 161k]
  ------------------
   87|   165k|            fp = dav1d_msac_decode_symbol_adapt4(msac, mv_comp->class0_fp[up], 3);
  ------------------
  |  |   47|   165k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
   88|   165k|            if (mv_prec > 0) // allow_high_precision_mv
  ------------------
  |  Branch (88:17): [True: 114k, False: 51.3k]
  ------------------
   89|   114k|                hp = dav1d_msac_decode_bool_adapt(msac, mv_comp->class0_hp);
  ------------------
  |  |   52|   114k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   90|   165k|        }
   91|  1.28M|    } else {
   92|  1.28M|        up = 1 << cl;
   93|  13.7M|        for (int n = 0; n < cl; n++)
  ------------------
  |  Branch (93:25): [True: 12.4M, False: 1.28M]
  ------------------
   94|  12.4M|            up |= dav1d_msac_decode_bool_adapt(msac, mv_comp->classN[n]) << n;
  ------------------
  |  |   52|  12.4M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   95|  1.28M|        if (mv_prec >= 0) {  // !force_integer_mv
  ------------------
  |  Branch (95:13): [True: 71.7k, False: 1.21M]
  ------------------
   96|  71.7k|            fp = dav1d_msac_decode_symbol_adapt4(msac, mv_comp->classN_fp, 3);
  ------------------
  |  |   47|  71.7k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
   97|  71.7k|            if (mv_prec > 0) // allow_high_precision_mv
  ------------------
  |  Branch (97:17): [True: 48.9k, False: 22.8k]
  ------------------
   98|  48.9k|                hp = dav1d_msac_decode_bool_adapt(msac, mv_comp->classN_hp);
  ------------------
  |  |   52|  48.9k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   99|  71.7k|        }
  100|  1.28M|    }
  101|       |
  102|  1.61M|    const int diff = ((up << 3) | (fp << 1) | hp) + 1;
  103|       |
  104|  1.61M|    return sign ? -diff : diff;
  ------------------
  |  Branch (104:12): [True: 1.40M, False: 208k]
  ------------------
  105|  1.61M|}
decode.c:read_vartx_tree:
  448|  1.72M|{
  449|  1.72M|    const Dav1dFrameContext *const f = t->f;
  450|  1.72M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  451|  1.72M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  452|       |
  453|       |    // var-tx tree coding
  454|  1.72M|    uint16_t tx_split[2] = { 0 };
  455|  1.72M|    b->max_ytx = dav1d_max_txfm_size_for_bs[bs][0];
  456|  1.72M|    if (!b->skip && (f->frame_hdr->segmentation.lossless[b->seg_id] ||
  ------------------
  |  Branch (456:9): [True: 459k, False: 1.26M]
  |  Branch (456:22): [True: 27.6k, False: 431k]
  ------------------
  457|   431k|                     b->max_ytx == TX_4X4))
  ------------------
  |  Branch (457:22): [True: 24.0k, False: 407k]
  ------------------
  458|  51.6k|    {
  459|  51.6k|        b->max_ytx = b->uvtx = TX_4X4;
  460|  51.6k|        if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
  ------------------
  |  Branch (460:13): [True: 11.4k, False: 40.2k]
  ------------------
  461|  11.4k|            dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], TX_4X4);
  462|  11.4k|            dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], TX_4X4);
  463|  11.4k|        }
  464|  1.67M|    } else if (f->frame_hdr->txfm_mode != DAV1D_TX_SWITCHABLE || b->skip) {
  ------------------
  |  Branch (464:16): [True: 1.28M, False: 396k]
  |  Branch (464:66): [True: 246k, False: 150k]
  ------------------
  465|  1.52M|        if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
  ------------------
  |  Branch (465:13): [True: 246k, False: 1.28M]
  ------------------
  466|   246k|            dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], b_dim[2 + 0]);
  467|   246k|            dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], b_dim[2 + 1]);
  468|   246k|        }
  469|  1.52M|        b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
  470|  1.52M|    } else {
  471|   150k|        assert(bw4 <= 16 || bh4 <= 16 || b->max_ytx == TX_64X64);
  ------------------
  |  Branch (471:9): [True: 147k, False: 3.18k]
  |  Branch (471:9): [True: 594, False: 2.59k]
  |  Branch (471:9): [True: 2.59k, False: 0]
  ------------------
  472|   150k|        int y, x, y_off, x_off;
  473|   150k|        const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
  474|   303k|        for (y = 0, y_off = 0; y < bh4; y += ytx->h, y_off++) {
  ------------------
  |  Branch (474:32): [True: 153k, False: 150k]
  ------------------
  475|   312k|            for (x = 0, x_off = 0; x < bw4; x += ytx->w, x_off++) {
  ------------------
  |  Branch (475:36): [True: 159k, False: 153k]
  ------------------
  476|   159k|                read_tx_tree(t, b->max_ytx, 0, tx_split, x_off, y_off);
  477|       |                // contexts are updated inside read_tx_tree()
  478|   159k|                t->bx += ytx->w;
  479|   159k|            }
  480|   153k|            t->bx -= x;
  481|   153k|            t->by += ytx->h;
  482|   153k|        }
  483|   150k|        t->by -= y;
  484|   150k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   150k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 150k]
  |  |  ------------------
  |  |   35|   150k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   150k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  485|      0|            printf("Post-vartxtree[%x/%x]: r=%d\n",
  486|      0|                   tx_split[0], tx_split[1], t->ts->msac.rng);
  487|   150k|        b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
  488|   150k|    }
  489|  1.72M|    assert(!(tx_split[0] & ~0x33));
  ------------------
  |  Branch (489:5): [True: 1.72M, False: 0]
  ------------------
  490|  1.72M|    b->tx_split0 = (uint8_t)tx_split[0];
  491|  1.72M|    b->tx_split1 = tx_split[1];
  492|  1.72M|}
decode.c:read_tx_tree:
  123|   326k|{
  124|   326k|    const Dav1dFrameContext *const f = t->f;
  125|   326k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  126|   326k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[from];
  127|   326k|    const int txw = t_dim->lw, txh = t_dim->lh;
  128|   326k|    int is_split;
  129|       |
  130|   326k|    if (depth < 2 && from > (int) TX_4X4) {
  ------------------
  |  Branch (130:9): [True: 274k, False: 51.9k]
  |  Branch (130:22): [True: 274k, False: 0]
  ------------------
  131|   274k|        const int cat = 2 * (TX_64X64 - t_dim->max) - depth;
  132|   274k|        const int a = t->a->tx[bx4] < txw;
  133|   274k|        const int l = t->l.tx[by4] < txh;
  134|       |
  135|   274k|        is_split = dav1d_msac_decode_bool_adapt(&t->ts->msac,
  ------------------
  |  |   52|   274k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  136|   274k|                       t->ts->cdf.m.txpart[cat][a + l]);
  137|   274k|        if (is_split)
  ------------------
  |  Branch (137:13): [True: 78.5k, False: 195k]
  ------------------
  138|  78.5k|            masks[depth] |= 1 << (y_off * 4 + x_off);
  139|   274k|    } else {
  140|  51.9k|        is_split = 0;
  141|  51.9k|    }
  142|       |
  143|   326k|    if (is_split && t_dim->max > TX_8X8) {
  ------------------
  |  Branch (143:9): [True: 78.5k, False: 247k]
  |  Branch (143:21): [True: 59.2k, False: 19.2k]
  ------------------
  144|  59.2k|        const enum RectTxfmSize sub = t_dim->sub;
  145|  59.2k|        const TxfmInfo *const sub_t_dim = &dav1d_txfm_dimensions[sub];
  146|  59.2k|        const int txsw = sub_t_dim->w, txsh = sub_t_dim->h;
  147|       |
  148|  59.2k|        read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 0, y_off * 2 + 0);
  149|  59.2k|        t->bx += txsw;
  150|  59.2k|        if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (150:13): [True: 45.8k, False: 13.4k]
  |  Branch (150:27): [True: 45.2k, False: 576]
  ------------------
  151|  45.2k|            read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 1, y_off * 2 + 0);
  152|  59.2k|        t->bx -= txsw;
  153|  59.2k|        t->by += txsh;
  154|  59.2k|        if (txh >= txw && t->by < f->bh) {
  ------------------
  |  Branch (154:13): [True: 39.5k, False: 19.7k]
  |  Branch (154:27): [True: 38.2k, False: 1.25k]
  ------------------
  155|  38.2k|            read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 0, y_off * 2 + 1);
  156|  38.2k|            t->bx += txsw;
  157|  38.2k|            if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (157:17): [True: 24.8k, False: 13.4k]
  |  Branch (157:31): [True: 24.3k, False: 558]
  ------------------
  158|  24.3k|                read_tx_tree(t, sub, depth + 1, masks,
  159|  24.3k|                             x_off * 2 + 1, y_off * 2 + 1);
  160|  38.2k|            t->bx -= txsw;
  161|  38.2k|        }
  162|  59.2k|        t->by -= txsh;
  163|   266k|    } else {
  164|   266k|        dav1d_memset_pow2[t_dim->lw](&t->a->tx[bx4], is_split ? TX_4X4 : txw);
  ------------------
  |  Branch (164:54): [True: 19.2k, False: 247k]
  ------------------
  165|   266k|        dav1d_memset_pow2[t_dim->lh](&t->l.tx[by4], is_split ? TX_4X4 : txh);
  ------------------
  |  Branch (165:53): [True: 19.2k, False: 247k]
  ------------------
  166|   266k|    }
  167|   326k|}
decode.c:splat_intrabc_mv:
  535|   669k|{
  536|   669k|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  537|   669k|        .ref.ref = { 0, -1 },
  538|   669k|        .mv.mv[0] = b->mv[0],
  539|   669k|        .bs = bs,
  540|   669k|        .mf = 0,
  541|   669k|    };
  542|   669k|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  543|   669k|}
decode.c:findoddzero:
  339|   446k|static inline int findoddzero(const uint8_t *buf, int len) {
  340|   478k|    for (int n = 0; n < len; n++)
  ------------------
  |  Branch (340:21): [True: 464k, False: 13.8k]
  ------------------
  341|   464k|        if (!buf[n * 2]) return 1;
  ------------------
  |  Branch (341:13): [True: 432k, False: 31.1k]
  ------------------
  342|  13.8k|    return 0;
  343|   446k|}
decode.c:find_matching_ref:
  197|   432k|{
  198|   432k|    /*const*/ refmvs_block *const *r = &t->rt.r[(t->by & 31) + 5];
  199|   432k|    int count = 0;
  200|   432k|    int have_topleft = have_top && have_left;
  ------------------
  |  Branch (200:24): [True: 388k, False: 44.1k]
  |  Branch (200:36): [True: 358k, False: 29.9k]
  ------------------
  201|   432k|    int have_topright = imax(bw4, bh4) < 32 &&
  ------------------
  |  Branch (201:25): [True: 402k, False: 30.6k]
  ------------------
  202|   402k|                        have_top && t->bx + bw4 < t->ts->tiling.col_end &&
  ------------------
  |  Branch (202:25): [True: 364k, False: 37.3k]
  |  Branch (202:37): [True: 339k, False: 25.0k]
  ------------------
  203|   339k|                        (intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  ------------------
  |  Branch (203:25): [True: 210k, False: 129k]
  ------------------
  204|       |
  205|   432k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  206|   432k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  207|       |
  208|   432k|    if (have_top) {
  ------------------
  |  Branch (208:9): [True: 388k, False: 44.1k]
  ------------------
  209|   388k|        const refmvs_block *r2 = &r[-1][t->bx];
  210|   388k|        if (matches(r2)) {
  ------------------
  |  |  206|   388k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 338k, False: 49.9k]
  |  |  |  Branch (206:53): [True: 315k, False: 23.5k]
  |  |  ------------------
  ------------------
  211|   315k|            masks[0] |= 1;
  212|   315k|            count = 1;
  213|   315k|        }
  214|   388k|        int aw4 = bs(r2)[0];
  ------------------
  |  |  205|   388k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  215|   388k|        if (aw4 >= bw4) {
  ------------------
  |  Branch (215:13): [True: 334k, False: 54.2k]
  ------------------
  216|   334k|            const int off = t->bx & (aw4 - 1);
  217|   334k|            if (off) have_topleft = 0;
  ------------------
  |  Branch (217:17): [True: 63.0k, False: 271k]
  ------------------
  218|   334k|            if (aw4 - off > bw4) have_topright = 0;
  ------------------
  |  Branch (218:17): [True: 64.2k, False: 270k]
  ------------------
  219|   334k|        } else {
  220|  54.2k|            unsigned mask = 1 << aw4;
  221|   137k|            for (int x = aw4; x < w4; x += aw4) {
  ------------------
  |  Branch (221:31): [True: 83.7k, False: 53.6k]
  ------------------
  222|  83.7k|                r2 += aw4;
  223|  83.7k|                if (matches(r2)) {
  ------------------
  |  |  206|  83.7k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 68.9k, False: 14.7k]
  |  |  |  Branch (206:53): [True: 65.3k, False: 3.66k]
  |  |  ------------------
  ------------------
  224|  65.3k|                    masks[0] |= mask;
  225|  65.3k|                    if (++count >= 8) return;
  ------------------
  |  Branch (225:25): [True: 598, False: 64.7k]
  ------------------
  226|  65.3k|                }
  227|  83.1k|                aw4 = bs(r2)[0];
  ------------------
  |  |  205|  83.1k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  228|  83.1k|                mask <<= aw4;
  229|  83.1k|            }
  230|  54.2k|        }
  231|   388k|    }
  232|   432k|    if (have_left) {
  ------------------
  |  Branch (232:9): [True: 402k, False: 29.8k]
  ------------------
  233|   402k|        /*const*/ refmvs_block *const *r2 = r;
  234|   402k|        if (matches(&r2[0][t->bx - 1])) {
  ------------------
  |  |  206|   402k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 349k, False: 52.5k]
  |  |  |  Branch (206:53): [True: 325k, False: 24.6k]
  |  |  ------------------
  ------------------
  235|   325k|            masks[1] |= 1;
  236|   325k|            if (++count >= 8) return;
  ------------------
  |  Branch (236:17): [True: 389, False: 324k]
  ------------------
  237|   325k|        }
  238|   402k|        int lh4 = bs(&r2[0][t->bx - 1])[1];
  ------------------
  |  |  205|   402k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  239|   402k|        if (lh4 >= bh4) {
  ------------------
  |  Branch (239:13): [True: 337k, False: 64.9k]
  ------------------
  240|   337k|            if (t->by & (lh4 - 1)) have_topleft = 0;
  ------------------
  |  Branch (240:17): [True: 64.5k, False: 272k]
  ------------------
  241|   337k|        } else {
  242|  64.9k|            unsigned mask = 1 << lh4;
  243|   160k|            for (int y = lh4; y < h4; y += lh4) {
  ------------------
  |  Branch (243:31): [True: 96.9k, False: 63.5k]
  ------------------
  244|  96.9k|                r2 += lh4;
  245|  96.9k|                if (matches(&r2[0][t->bx - 1])) {
  ------------------
  |  |  206|  96.9k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 73.3k, False: 23.6k]
  |  |  |  Branch (206:53): [True: 69.0k, False: 4.25k]
  |  |  ------------------
  ------------------
  246|  69.0k|                    masks[1] |= mask;
  247|  69.0k|                    if (++count >= 8) return;
  ------------------
  |  Branch (247:25): [True: 1.41k, False: 67.6k]
  ------------------
  248|  69.0k|                }
  249|  95.5k|                lh4 = bs(&r2[0][t->bx - 1])[1];
  ------------------
  |  |  205|  95.5k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  250|  95.5k|                mask <<= lh4;
  251|  95.5k|            }
  252|  64.9k|        }
  253|   402k|    }
  254|   430k|    if (have_topleft && matches(&r[-1][t->bx - 1])) {
  ------------------
  |  |  206|   229k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 189k, False: 39.2k]
  |  |  |  Branch (206:53): [True: 175k, False: 14.0k]
  |  |  ------------------
  ------------------
  |  Branch (254:9): [True: 229k, False: 201k]
  ------------------
  255|   175k|        masks[1] |= 1ULL << 32;
  256|   175k|        if (++count >= 8) return;
  ------------------
  |  Branch (256:13): [True: 1.04k, False: 174k]
  ------------------
  257|   175k|    }
  258|   429k|    if (have_topright && matches(&r[-1][t->bx + bw4])) {
  ------------------
  |  |  206|   145k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 118k, False: 26.8k]
  |  |  |  Branch (206:53): [True: 110k, False: 8.38k]
  |  |  ------------------
  ------------------
  |  Branch (258:9): [True: 145k, False: 283k]
  ------------------
  259|   110k|        masks[0] |= 1ULL << 32;
  260|   110k|    }
  261|   429k|#undef matches
  262|   429k|}
decode.c:derive_warpmv:
  268|  79.8k|{
  269|  79.8k|    int pts[8][2 /* in, out */][2 /* x, y */], np = 0;
  270|  79.8k|    /*const*/ refmvs_block *const *r = &t->rt.r[(t->by & 31) + 5];
  271|       |
  272|  79.8k|#define add_sample(dx, dy, sx, sy, rp) do { \
  273|  79.8k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  274|  79.8k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  275|  79.8k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  276|  79.8k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  277|  79.8k|    np++; \
  278|  79.8k|} while (0)
  279|       |
  280|       |    // use masks[] to find the projectable motion vectors in the edges
  281|  79.8k|    if ((unsigned) masks[0] == 1 && !(masks[1] >> 32)) {
  ------------------
  |  Branch (281:9): [True: 58.0k, False: 21.7k]
  |  Branch (281:37): [True: 24.2k, False: 33.8k]
  ------------------
  282|  24.2k|        const int off = t->bx & (bs(&r[-1][t->bx])[0] - 1);
  ------------------
  |  |  205|  24.2k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  283|  24.2k|        add_sample(-off, 0, 1, -1, &r[-1][t->bx]);
  ------------------
  |  |  272|  24.2k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  24.2k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  24.2k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  24.2k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  24.2k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  24.2k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  24.2k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  24.2k|    np++; \
  |  |  278|  24.2k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 24.2k]
  |  |  ------------------
  ------------------
  284|   109k|    } else for (unsigned off = 0, xmask = (uint32_t) masks[0]; np < 8 && xmask;) { // top
  ------------------
  |  Branch (284:64): [True: 109k, False: 295]
  |  Branch (284:74): [True: 53.7k, False: 55.3k]
  ------------------
  285|  53.7k|        const int tz = ctz(xmask);
  286|  53.7k|        off += tz;
  287|  53.7k|        xmask >>= tz;
  288|  53.7k|        add_sample(off, 0, 1, -1, &r[-1][t->bx + off]);
  ------------------
  |  |  272|  53.7k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  53.7k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  53.7k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  53.7k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  53.7k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  53.7k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  53.7k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  53.7k|    np++; \
  |  |  278|  53.7k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 53.7k]
  |  |  ------------------
  ------------------
  289|  53.7k|        xmask &= ~1;
  290|  53.7k|    }
  291|  79.8k|    if (np < 8 && masks[1] == 1) {
  ------------------
  |  Branch (291:9): [True: 79.5k, False: 295]
  |  Branch (291:19): [True: 29.3k, False: 50.2k]
  ------------------
  292|  29.3k|        const int off = t->by & (bs(&r[0][t->bx - 1])[1] - 1);
  ------------------
  |  |  205|  29.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  293|  29.3k|        add_sample(0, -off, -1, 1, &r[-off][t->bx - 1]);
  ------------------
  |  |  272|  29.3k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  29.3k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  29.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  29.3k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  29.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  29.3k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  29.3k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  29.3k|    np++; \
  |  |  278|  29.3k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 29.3k]
  |  |  ------------------
  ------------------
  294|   107k|    } else for (unsigned off = 0, ymask = (uint32_t) masks[1]; np < 8 && ymask;) { // left
  ------------------
  |  Branch (294:64): [True: 106k, False: 816]
  |  Branch (294:74): [True: 57.0k, False: 49.6k]
  ------------------
  295|  57.0k|        const int tz = ctz(ymask);
  296|  57.0k|        off += tz;
  297|  57.0k|        ymask >>= tz;
  298|  57.0k|        add_sample(0, off, -1, 1, &r[off][t->bx - 1]);
  ------------------
  |  |  272|  57.0k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  57.0k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  57.0k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  57.0k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  57.0k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  57.0k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  57.0k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  57.0k|    np++; \
  |  |  278|  57.0k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 57.0k]
  |  |  ------------------
  ------------------
  299|  57.0k|        ymask &= ~1;
  300|  57.0k|    }
  301|  79.8k|    if (np < 8 && masks[1] >> 32) // top/left
  ------------------
  |  Branch (301:9): [True: 78.8k, False: 1.01k]
  |  Branch (301:19): [True: 40.3k, False: 38.4k]
  ------------------
  302|  40.3k|        add_sample(0, 0, -1, -1, &r[-1][t->bx - 1]);
  ------------------
  |  |  272|  40.3k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  40.3k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  40.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  40.3k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  40.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  40.3k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  40.3k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  40.3k|    np++; \
  |  |  278|  40.3k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 40.3k]
  |  |  ------------------
  ------------------
  303|  79.8k|    if (np < 8 && masks[0] >> 32) // top/right
  ------------------
  |  Branch (303:9): [True: 78.4k, False: 1.35k]
  |  Branch (303:19): [True: 27.3k, False: 51.1k]
  ------------------
  304|  27.3k|        add_sample(bw4, 0, 1, -1, &r[-1][t->bx + bw4]);
  ------------------
  |  |  272|  27.3k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  27.3k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  27.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  27.3k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  27.3k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  27.3k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  27.3k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  27.3k|    np++; \
  |  |  278|  27.3k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 27.3k]
  |  |  ------------------
  ------------------
  305|  79.8k|    assert(np > 0 && np <= 8);
  ------------------
  |  Branch (305:5): [True: 79.8k, False: 0]
  |  Branch (305:5): [True: 79.8k, False: 0]
  ------------------
  306|  79.8k|#undef bs
  307|       |
  308|       |    // select according to motion vector difference against a threshold
  309|  79.8k|    int mvd[8], ret = 0;
  310|  79.8k|    const int thresh = 4 * iclip(imax(bw4, bh4), 4, 28);
  311|   311k|    for (int i = 0; i < np; i++) {
  ------------------
  |  Branch (311:21): [True: 232k, False: 79.8k]
  ------------------
  312|   232k|        mvd[i] = abs(pts[i][1][0] - pts[i][0][0] - mv.x) +
  313|   232k|                 abs(pts[i][1][1] - pts[i][0][1] - mv.y);
  314|   232k|        if (mvd[i] > thresh)
  ------------------
  |  Branch (314:13): [True: 50.2k, False: 181k]
  ------------------
  315|  50.2k|            mvd[i] = -1;
  316|   181k|        else
  317|   181k|            ret++;
  318|   232k|    }
  319|  79.8k|    if (!ret) {
  ------------------
  |  Branch (319:9): [True: 11.2k, False: 68.5k]
  ------------------
  320|  11.2k|        ret = 1;
  321|  77.6k|    } else for (int i = 0, j = np - 1, k = 0; k < np - ret; k++, i++, j--) {
  ------------------
  |  Branch (321:47): [True: 20.4k, False: 57.2k]
  ------------------
  322|  31.5k|        while (mvd[i] != -1) i++;
  ------------------
  |  Branch (322:16): [True: 11.1k, False: 20.4k]
  ------------------
  323|  39.5k|        while (mvd[j] == -1) j--;
  ------------------
  |  Branch (323:16): [True: 19.1k, False: 20.4k]
  ------------------
  324|  20.4k|        assert(i != j);
  ------------------
  |  Branch (324:9): [True: 20.4k, False: 0]
  ------------------
  325|  20.4k|        if (i > j) break;
  ------------------
  |  Branch (325:13): [True: 11.3k, False: 9.09k]
  ------------------
  326|       |        // replace the discarded samples;
  327|  9.09k|        mvd[i] = mvd[j];
  328|  9.09k|        memcpy(pts[i], pts[j], sizeof(*pts));
  329|  9.09k|    }
  330|       |
  331|  79.8k|    if (!dav1d_find_affine_int(pts, ret, bw4, bh4, mv, wmp, t->bx, t->by) &&
  ------------------
  |  Branch (331:9): [True: 76.3k, False: 3.46k]
  ------------------
  332|  76.3k|        !dav1d_get_shear_params(wmp))
  ------------------
  |  Branch (332:9): [True: 73.1k, False: 3.17k]
  ------------------
  333|  73.1k|    {
  334|  73.1k|        wmp->type = DAV1D_WM_TYPE_AFFINE;
  335|  73.1k|    } else
  336|  6.63k|        wmp->type = DAV1D_WM_TYPE_IDENTITY;
  337|  79.8k|}
decode.c:splat_tworef_mv:
  550|   168k|{
  551|   168k|    assert(bw4 >= 2 && bh4 >= 2);
  ------------------
  |  Branch (551:5): [True: 168k, False: 0]
  |  Branch (551:5): [True: 168k, False: 0]
  ------------------
  552|   168k|    const enum CompInterPredMode mode = b->inter_mode;
  553|   168k|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  554|   168k|        .ref.ref = { b->ref[0] + 1, b->ref[1] + 1 },
  555|   168k|        .mv.mv = { b->mv[0], b->mv[1] },
  556|   168k|        .bs = bs,
  557|   168k|        .mf = (mode == GLOBALMV_GLOBALMV) | !!((1 << mode) & (0xbc)) * 2,
  558|   168k|    };
  559|   168k|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  560|   168k|}
decode.c:splat_oneref_mv:
  519|   890k|{
  520|   890k|    const enum InterPredMode mode = b->inter_mode;
  521|   890k|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  522|   890k|        .ref.ref = { b->ref[0] + 1, b->interintra_type ? 0 : -1 },
  ------------------
  |  Branch (522:37): [True: 41.5k, False: 848k]
  ------------------
  523|   890k|        .mv.mv[0] = b->mv[0],
  524|   890k|        .bs = bs,
  525|   890k|        .mf = (mode == GLOBALMV && imin(bw4, bh4) >= 2) | ((mode == NEWMV) * 2),
  ------------------
  |  Branch (525:16): [True: 410k, False: 479k]
  |  Branch (525:36): [True: 285k, False: 124k]
  ------------------
  526|   890k|    };
  527|   890k|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  528|   890k|}
decode.c:read_restoration_info:
 2516|   108k|{
 2517|   108k|    const Dav1dFrameContext *const f = t->f;
 2518|   108k|    Dav1dTileState *const ts = t->ts;
 2519|   108k|    const Av1RestorationUnit *const lr_ref = ts->lr_ref[p];
 2520|       |
 2521|   108k|    if (frame_type == DAV1D_RESTORATION_SWITCHABLE) {
  ------------------
  |  Branch (2521:9): [True: 39.3k, False: 69.6k]
  ------------------
 2522|  39.3k|        const int filter = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|  39.3k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 2523|  39.3k|                               ts->cdf.m.restore_switchable, 2);
 2524|  39.3k|        lr->type = filter + !!filter; /* NONE/WIENER/SGRPROJ */
 2525|  69.6k|    } else {
 2526|  69.6k|        const unsigned type =
 2527|  69.6k|            dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  69.6k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 2528|  69.6k|                frame_type == DAV1D_RESTORATION_WIENER ?
  ------------------
  |  Branch (2528:17): [True: 22.3k, False: 47.2k]
  ------------------
 2529|  47.2k|                ts->cdf.m.restore_wiener : ts->cdf.m.restore_sgrproj);
 2530|  69.6k|        lr->type = type ? frame_type : DAV1D_RESTORATION_NONE;
  ------------------
  |  Branch (2530:20): [True: 34.7k, False: 34.9k]
  ------------------
 2531|  69.6k|    }
 2532|       |
 2533|   108k|    if (lr->type == DAV1D_RESTORATION_WIENER) {
  ------------------
  |  Branch (2533:9): [True: 16.6k, False: 92.3k]
  ------------------
 2534|  16.6k|        lr->filter_v[0] = p ? 0 :
  ------------------
  |  Branch (2534:27): [True: 9.97k, False: 6.64k]
  ------------------
 2535|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2536|  6.64k|                lr_ref->filter_v[0] + 5, 16, 1) - 5;
 2537|  16.6k|        lr->filter_v[1] =
 2538|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2539|  16.6k|                lr_ref->filter_v[1] + 23, 32, 2) - 23;
 2540|  16.6k|        lr->filter_v[2] =
 2541|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2542|  16.6k|                lr_ref->filter_v[2] + 17, 64, 3) - 17;
 2543|       |
 2544|  16.6k|        lr->filter_h[0] = p ? 0 :
  ------------------
  |  Branch (2544:27): [True: 9.97k, False: 6.64k]
  ------------------
 2545|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2546|  6.64k|                lr_ref->filter_h[0] + 5, 16, 1) - 5;
 2547|  16.6k|        lr->filter_h[1] =
 2548|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2549|  16.6k|                lr_ref->filter_h[1] + 23, 32, 2) - 23;
 2550|  16.6k|        lr->filter_h[2] =
 2551|  16.6k|            dav1d_msac_decode_subexp(&ts->msac,
 2552|  16.6k|                lr_ref->filter_h[2] + 17, 64, 3) - 17;
 2553|  16.6k|        memcpy(lr->sgr_weights, lr_ref->sgr_weights, sizeof(lr->sgr_weights));
 2554|  16.6k|        ts->lr_ref[p] = lr;
 2555|  16.6k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  16.6k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 16.6k]
  |  |  ------------------
  |  |   35|  16.6k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  16.6k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2556|      0|            printf("Post-lr_wiener[pl=%d,v[%d,%d,%d],h[%d,%d,%d]]: r=%d\n",
 2557|      0|                   p, lr->filter_v[0], lr->filter_v[1],
 2558|      0|                   lr->filter_v[2], lr->filter_h[0],
 2559|      0|                   lr->filter_h[1], lr->filter_h[2], ts->msac.rng);
 2560|  92.3k|    } else if (lr->type == DAV1D_RESTORATION_SGRPROJ) {
  ------------------
  |  Branch (2560:16): [True: 36.0k, False: 56.3k]
  ------------------
 2561|  36.0k|        const unsigned idx = dav1d_msac_decode_bools(&ts->msac, 4);
 2562|  36.0k|        const uint16_t *const sgr_params = dav1d_sgr_params[idx];
 2563|  36.0k|        lr->type += idx;
 2564|  36.0k|        lr->sgr_weights[0] = sgr_params[0] ? dav1d_msac_decode_subexp(&ts->msac,
  ------------------
  |  Branch (2564:30): [True: 27.9k, False: 8.09k]
  ------------------
 2565|  27.9k|            lr_ref->sgr_weights[0] + 96, 128, 4) - 96 : 0;
 2566|  36.0k|        lr->sgr_weights[1] = sgr_params[1] ? dav1d_msac_decode_subexp(&ts->msac,
  ------------------
  |  Branch (2566:30): [True: 26.1k, False: 9.88k]
  ------------------
 2567|  26.1k|            lr_ref->sgr_weights[1] + 32, 128, 4) - 32 : 95;
 2568|  36.0k|        memcpy(lr->filter_v, lr_ref->filter_v, sizeof(lr->filter_v));
 2569|  36.0k|        memcpy(lr->filter_h, lr_ref->filter_h, sizeof(lr->filter_h));
 2570|  36.0k|        ts->lr_ref[p] = lr;
 2571|  36.0k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  36.0k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 36.0k]
  |  |  ------------------
  |  |   35|  36.0k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  36.0k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2572|      0|            printf("Post-lr_sgrproj[pl=%d,idx=%d,w[%d,%d]]: r=%d\n",
 2573|      0|                   p, idx, lr->sgr_weights[0],
 2574|      0|                   lr->sgr_weights[1], ts->msac.rng);
 2575|  36.0k|    }
 2576|   108k|}
decode.c:init_quant_tables:
   57|  56.8k|{
   58|   204k|    for (int i = 0; i < (frame_hdr->segmentation.enabled ? 8 : 1); i++) {
  ------------------
  |  Branch (58:21): [True: 147k, False: 56.8k]
  |  Branch (58:26): [True: 116k, False: 87.7k]
  ------------------
   59|   147k|        const int yac = frame_hdr->segmentation.enabled ?
  ------------------
  |  Branch (59:25): [True: 103k, False: 43.8k]
  ------------------
   60|   103k|            iclip_u8(qidx + frame_hdr->segmentation.seg_data.d[i].delta_q) : qidx;
   61|   147k|        const int ydc = iclip_u8(yac + frame_hdr->quant.ydc_delta);
   62|   147k|        const int uac = iclip_u8(yac + frame_hdr->quant.uac_delta);
   63|   147k|        const int udc = iclip_u8(yac + frame_hdr->quant.udc_delta);
   64|   147k|        const int vac = iclip_u8(yac + frame_hdr->quant.vac_delta);
   65|   147k|        const int vdc = iclip_u8(yac + frame_hdr->quant.vdc_delta);
   66|       |
   67|   147k|        dq[i][0][0] = dav1d_dq_tbl[seq_hdr->hbd][ydc][0];
   68|   147k|        dq[i][0][1] = dav1d_dq_tbl[seq_hdr->hbd][yac][1];
   69|   147k|        dq[i][1][0] = dav1d_dq_tbl[seq_hdr->hbd][udc][0];
   70|   147k|        dq[i][1][1] = dav1d_dq_tbl[seq_hdr->hbd][uac][1];
   71|   147k|        dq[i][2][0] = dav1d_dq_tbl[seq_hdr->hbd][vdc][0];
   72|   147k|        dq[i][2][1] = dav1d_dq_tbl[seq_hdr->hbd][vac][1];
   73|   147k|    }
   74|  56.8k|}
decode.c:setup_tile:
 2432|  50.5k|{
 2433|  50.5k|    const int col_sb_start = f->frame_hdr->tiling.col_start_sb[tile_col];
 2434|  50.5k|    const int col_sb128_start = col_sb_start >> !f->seq_hdr->sb128;
 2435|  50.5k|    const int col_sb_end = f->frame_hdr->tiling.col_start_sb[tile_col + 1];
 2436|  50.5k|    const int row_sb_start = f->frame_hdr->tiling.row_start_sb[tile_row];
 2437|  50.5k|    const int row_sb_end = f->frame_hdr->tiling.row_start_sb[tile_row + 1];
 2438|  50.5k|    const int sb_shift = f->sb_shift;
 2439|       |
 2440|  50.5k|    const uint8_t *const size_mul = ss_size_mul[f->cur.p.layout];
 2441|   151k|    for (int p = 0; p < 2; p++) {
  ------------------
  |  Branch (2441:21): [True: 101k, False: 50.5k]
  ------------------
 2442|   101k|        ts->frame_thread[p].pal_idx = f->frame_thread.pal_idx ?
  ------------------
  |  Branch (2442:39): [True: 0, False: 101k]
  ------------------
 2443|      0|            &f->frame_thread.pal_idx[(size_t)tile_start_off * size_mul[1] / 8] :
 2444|   101k|            NULL;
 2445|   101k|        ts->frame_thread[p].cbi = f->frame_thread.cbi ?
  ------------------
  |  Branch (2445:35): [True: 0, False: 101k]
  ------------------
 2446|      0|            &f->frame_thread.cbi[(size_t)tile_start_off * size_mul[0] / 64] :
 2447|   101k|            NULL;
 2448|   101k|        ts->frame_thread[p].cf = f->frame_thread.cf ?
  ------------------
  |  Branch (2448:34): [True: 0, False: 101k]
  ------------------
 2449|      0|            (uint8_t*)f->frame_thread.cf +
 2450|      0|                (((size_t)tile_start_off * size_mul[0]) >> !f->seq_hdr->hbd) :
 2451|   101k|            NULL;
 2452|   101k|    }
 2453|       |
 2454|  50.5k|    dav1d_cdf_thread_copy(&ts->cdf, &f->in_cdf);
 2455|  50.5k|    ts->last_qidx = f->frame_hdr->quant.yac;
 2456|  50.5k|    ts->last_delta_lf.u32 = 0;
 2457|       |
 2458|  50.5k|    dav1d_msac_init(&ts->msac, data, sz, f->frame_hdr->disable_cdf_update);
 2459|       |
 2460|  50.5k|    ts->tiling.row = tile_row;
 2461|  50.5k|    ts->tiling.col = tile_col;
 2462|  50.5k|    ts->tiling.col_start = col_sb_start << sb_shift;
 2463|  50.5k|    ts->tiling.col_end = imin(col_sb_end << sb_shift, f->bw);
 2464|  50.5k|    ts->tiling.row_start = row_sb_start << sb_shift;
 2465|  50.5k|    ts->tiling.row_end = imin(row_sb_end << sb_shift, f->bh);
 2466|       |
 2467|       |    // Reference Restoration Unit (used for exp coding)
 2468|  50.5k|    int sb_idx, unit_idx;
 2469|  50.5k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2469:9): [True: 5.60k, False: 44.8k]
  ------------------
 2470|       |        // vertical components only
 2471|  5.60k|        sb_idx = (ts->tiling.row_start >> 5) * f->sr_sb128w;
 2472|  5.60k|        unit_idx = (ts->tiling.row_start & 16) >> 3;
 2473|  44.8k|    } else {
 2474|  44.8k|        sb_idx = (ts->tiling.row_start >> 5) * f->sb128w + col_sb128_start;
 2475|  44.8k|        unit_idx = ((ts->tiling.row_start & 16) >> 3) +
 2476|  44.8k|                   ((ts->tiling.col_start & 16) >> 4);
 2477|  44.8k|    }
 2478|   202k|    for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (2478:21): [True: 151k, False: 50.5k]
  ------------------
 2479|   151k|        if (!((f->lf.restore_planes >> p) & 1U))
  ------------------
  |  Branch (2479:13): [True: 130k, False: 21.2k]
  ------------------
 2480|   130k|            continue;
 2481|       |
 2482|  21.2k|        if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2482:13): [True: 6.73k, False: 14.4k]
  ------------------
 2483|  6.73k|            const int ss_hor = p && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2483:32): [True: 3.81k, False: 2.92k]
  |  Branch (2483:37): [True: 2.60k, False: 1.21k]
  ------------------
 2484|  6.73k|            const int d = f->frame_hdr->super_res.width_scale_denominator;
 2485|  6.73k|            const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!p];
 2486|  6.73k|            const int rnd = (8 << unit_size_log2) - 1, shift = unit_size_log2 + 3;
 2487|  6.73k|            const int x = ((4 * ts->tiling.col_start * d >> ss_hor) + rnd) >> shift;
 2488|  6.73k|            const int px_x = x << (unit_size_log2 + ss_hor);
 2489|  6.73k|            const int u_idx = unit_idx + ((px_x & 64) >> 6);
 2490|  6.73k|            const int sb128x = px_x >> 7;
 2491|  6.73k|            if (sb128x >= f->sr_sb128w) continue;
  ------------------
  |  Branch (2491:17): [True: 359, False: 6.37k]
  ------------------
 2492|  6.37k|            ts->lr_ref[p] = &f->lf.lr_mask[sb_idx + sb128x].lr[p][u_idx];
 2493|  14.4k|        } else {
 2494|  14.4k|            ts->lr_ref[p] = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2495|  14.4k|        }
 2496|       |
 2497|  20.8k|        ts->lr_ref[p]->filter_v[0] = 3;
 2498|  20.8k|        ts->lr_ref[p]->filter_v[1] = -7;
 2499|  20.8k|        ts->lr_ref[p]->filter_v[2] = 15;
 2500|  20.8k|        ts->lr_ref[p]->filter_h[0] = 3;
 2501|  20.8k|        ts->lr_ref[p]->filter_h[1] = -7;
 2502|  20.8k|        ts->lr_ref[p]->filter_h[2] = 15;
 2503|  20.8k|        ts->lr_ref[p]->sgr_weights[0] = -32;
 2504|  20.8k|        ts->lr_ref[p]->sgr_weights[1] = 31;
 2505|  20.8k|    }
 2506|       |
 2507|  50.5k|    if (f->c->n_tc > 1) {
  ------------------
  |  Branch (2507:9): [True: 0, False: 50.5k]
  ------------------
 2508|      0|        for (int p = 0; p < 2; p++)
  ------------------
  |  Branch (2508:25): [True: 0, False: 0]
  ------------------
 2509|      0|            atomic_init(&ts->progress[p], row_sb_start);
 2510|      0|    }
 2511|  50.5k|}
decode.c:get_upscale_x0:
 3324|  9.66k|static int get_upscale_x0(const int in_w, const int out_w, const int step) {
 3325|  9.66k|    const int err = out_w * step - (in_w << 14);
 3326|  9.66k|    const int x0 = (-((out_w - in_w) << 13) + (out_w >> 1)) / out_w + 128 - (err / 2);
 3327|  9.66k|    return x0 & 0x3fff;
 3328|  9.66k|}

obu.c:get_poc_diff:
  239|   255k|{
  240|   255k|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 0, False: 255k]
  ------------------
  241|   255k|    const int mask = 1 << (order_hint_n_bits - 1);
  242|   255k|    const int diff = poc0 - poc1;
  243|   255k|    return (diff & (mask - 1)) - (diff & mask);
  244|   255k|}
refmvs.c:get_gmv_2d:
  482|  1.22M|{
  483|  1.22M|    switch (gmv->type) {
  484|   182k|    case DAV1D_WM_TYPE_ROT_ZOOM:
  ------------------
  |  Branch (484:5): [True: 182k, False: 1.04M]
  ------------------
  485|   182k|        assert(gmv->matrix[5] ==  gmv->matrix[2]);
  ------------------
  |  Branch (485:9): [True: 182k, False: 0]
  ------------------
  486|   182k|        assert(gmv->matrix[4] == -gmv->matrix[3]);
  ------------------
  |  Branch (486:9): [True: 182k, False: 0]
  ------------------
  487|       |        // fall-through
  488|   182k|    default:
  ------------------
  |  Branch (488:5): [True: 0, False: 1.22M]
  ------------------
  489|   224k|    case DAV1D_WM_TYPE_AFFINE: {
  ------------------
  |  Branch (489:5): [True: 42.1k, False: 1.18M]
  ------------------
  490|   224k|        const int x = bx4 * 4 + bw4 * 2 - 1;
  491|   224k|        const int y = by4 * 4 + bh4 * 2 - 1;
  492|   224k|        const int xc = (gmv->matrix[2] - (1 << 16)) * x +
  493|   224k|                       gmv->matrix[3] * y + gmv->matrix[0];
  494|   224k|        const int yc = (gmv->matrix[5] - (1 << 16)) * y +
  495|   224k|                       gmv->matrix[4] * x + gmv->matrix[1];
  496|   224k|        const int shift = 16 - (3 - !hdr->hp);
  497|   224k|        const int round = (1 << shift) >> 1;
  498|   224k|        mv res = (mv) {
  499|   224k|            .y = apply_sign(((abs(yc) + round) >> shift) << !hdr->hp, yc),
  500|   224k|            .x = apply_sign(((abs(xc) + round) >> shift) << !hdr->hp, xc),
  501|   224k|        };
  502|   224k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (502:13): [True: 70.4k, False: 154k]
  ------------------
  503|  70.4k|            fix_int_mv_precision(&res);
  504|   224k|        return res;
  505|   182k|    }
  506|   161k|    case DAV1D_WM_TYPE_TRANSLATION: {
  ------------------
  |  Branch (506:5): [True: 161k, False: 1.06M]
  ------------------
  507|   161k|        mv res = (mv) {
  508|   161k|            .y = gmv->matrix[0] >> 13,
  509|   161k|            .x = gmv->matrix[1] >> 13,
  510|   161k|        };
  511|   161k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (511:13): [True: 14.5k, False: 147k]
  ------------------
  512|  14.5k|            fix_int_mv_precision(&res);
  513|   161k|        return res;
  514|   182k|    }
  515|   841k|    case DAV1D_WM_TYPE_IDENTITY:
  ------------------
  |  Branch (515:5): [True: 841k, False: 386k]
  ------------------
  516|   841k|        return (mv) { .x = 0, .y = 0 };
  517|  1.22M|    }
  518|  1.22M|}
refmvs.c:fix_int_mv_precision:
  462|  97.4k|static inline void fix_int_mv_precision(mv *const mv) {
  463|  97.4k|    mv->x = (mv->x - (mv->x >> 15) + 3) & ~7U;
  464|  97.4k|    mv->y = (mv->y - (mv->y >> 15) + 3) & ~7U;
  465|  97.4k|}
refmvs.c:fix_mv_precision:
  469|   570k|{
  470|   570k|    if (hdr->force_integer_mv) {
  ------------------
  |  Branch (470:9): [True: 12.3k, False: 558k]
  ------------------
  471|  12.3k|        fix_int_mv_precision(mv);
  472|   558k|    } else if (!hdr->hp) {
  ------------------
  |  Branch (472:16): [True: 32.5k, False: 525k]
  ------------------
  473|  32.5k|        mv->x = (mv->x - (mv->x >> 15)) & ~1U;
  474|  32.5k|        mv->y = (mv->y - (mv->y >> 15)) & ~1U;
  475|  32.5k|    }
  476|   570k|}
refmvs.c:get_poc_diff:
  239|   556k|{
  240|   556k|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 240k, False: 316k]
  ------------------
  241|   316k|    const int mask = 1 << (order_hint_n_bits - 1);
  242|   316k|    const int diff = poc0 - poc1;
  243|   316k|    return (diff & (mask - 1)) - (diff & mask);
  244|   556k|}
decode.c:get_partition_ctx:
   87|  2.89M|{
   88|  2.89M|    return ((a->partition[xb8] >> (4 - bl)) & 1) +
   89|  2.89M|          (((l->partition[yb8] >> (4 - bl)) & 1) << 1);
   90|  2.89M|}
decode.c:get_cur_frame_segid:
  445|  1.04M|{
  446|  1.04M|    cur_seg_map += bx + by * stride;
  447|  1.04M|    if (have_left && have_top) {
  ------------------
  |  Branch (447:9): [True: 928k, False: 120k]
  |  Branch (447:22): [True: 664k, False: 264k]
  ------------------
  448|   664k|        const int l = cur_seg_map[-1];
  449|   664k|        const int a = cur_seg_map[-stride];
  450|   664k|        const int al = cur_seg_map[-(stride + 1)];
  451|       |
  452|   664k|        if (l == a && al == l) *seg_ctx = 2;
  ------------------
  |  Branch (452:13): [True: 402k, False: 262k]
  |  Branch (452:23): [True: 387k, False: 14.5k]
  ------------------
  453|   276k|        else if (l == a || al == l || a == al) *seg_ctx = 1;
  ------------------
  |  Branch (453:18): [True: 14.5k, False: 262k]
  |  Branch (453:28): [True: 113k, False: 148k]
  |  Branch (453:39): [True: 92.0k, False: 56.9k]
  ------------------
  454|  56.9k|        else *seg_ctx = 0;
  455|   664k|        return a == al ? a : l;
  ------------------
  |  Branch (455:16): [True: 479k, False: 184k]
  ------------------
  456|   664k|    } else {
  457|   384k|        *seg_ctx = 0;
  458|   384k|        return have_left ? cur_seg_map[-1] : have_top ? cur_seg_map[-stride] : 0;
  ------------------
  |  Branch (458:16): [True: 264k, False: 120k]
  |  Branch (458:46): [True: 116k, False: 3.72k]
  ------------------
  459|   384k|    }
  460|  1.04M|}
decode.c:get_intra_ctx:
   63|   775k|{
   64|   775k|    if (have_left) {
  ------------------
  |  Branch (64:9): [True: 739k, False: 36.6k]
  ------------------
   65|   739k|        if (have_top) {
  ------------------
  |  Branch (65:13): [True: 646k, False: 92.8k]
  ------------------
   66|   646k|            const int ctx = l->intra[yb4] + a->intra[xb4];
   67|   646k|            return ctx + (ctx == 2);
   68|   646k|        } else
   69|  92.8k|            return l->intra[yb4] * 2;
   70|   739k|    } else {
   71|  36.6k|        return have_top ? a->intra[xb4] * 2 : 0;
  ------------------
  |  Branch (71:16): [True: 24.0k, False: 12.5k]
  ------------------
   72|  36.6k|    }
   73|   775k|}
decode.c:get_tx_ctx:
   79|   590k|{
   80|   590k|    return (l->tx_intra[yb4] >= max_tx->lh) + (a->tx_intra[xb4] >= max_tx->lw);
   81|   590k|}
decode.c:get_comp_ctx:
  160|   276k|{
  161|   276k|    if (have_top) {
  ------------------
  |  Branch (161:9): [True: 234k, False: 42.1k]
  ------------------
  162|   234k|        if (have_left) {
  ------------------
  |  Branch (162:13): [True: 223k, False: 10.5k]
  ------------------
  163|   223k|            if (a->comp_type[xb4]) {
  ------------------
  |  Branch (163:17): [True: 103k, False: 120k]
  ------------------
  164|   103k|                if (l->comp_type[yb4]) {
  ------------------
  |  Branch (164:21): [True: 75.4k, False: 28.0k]
  ------------------
  165|  75.4k|                    return 4;
  166|  75.4k|                } else {
  167|       |                    // 4U means intra (-1) or bwd (>= 4)
  168|  28.0k|                    return 2 + ((unsigned)l->ref[0][yb4] >= 4U);
  169|  28.0k|                }
  170|   120k|            } else if (l->comp_type[yb4]) {
  ------------------
  |  Branch (170:24): [True: 34.6k, False: 85.7k]
  ------------------
  171|       |                // 4U means intra (-1) or bwd (>= 4)
  172|  34.6k|                return 2 + ((unsigned)a->ref[0][xb4] >= 4U);
  173|  85.7k|            } else {
  174|  85.7k|                return (l->ref[0][yb4] >= 4) ^ (a->ref[0][xb4] >= 4);
  175|  85.7k|            }
  176|   223k|        } else {
  177|  10.5k|            return a->comp_type[xb4] ? 3 : a->ref[0][xb4] >= 4;
  ------------------
  |  Branch (177:20): [True: 5.18k, False: 5.37k]
  ------------------
  178|  10.5k|        }
  179|   234k|    } else if (have_left) {
  ------------------
  |  Branch (179:16): [True: 38.5k, False: 3.58k]
  ------------------
  180|  38.5k|        return l->comp_type[yb4] ? 3 : l->ref[0][yb4] >= 4;
  ------------------
  |  Branch (180:16): [True: 18.7k, False: 19.7k]
  ------------------
  181|  38.5k|    } else {
  182|  3.58k|        return 1;
  183|  3.58k|    }
  184|   276k|}
decode.c:fix_mv_precision:
  469|   534k|{
  470|   534k|    if (hdr->force_integer_mv) {
  ------------------
  |  Branch (470:9): [True: 146k, False: 387k]
  ------------------
  471|   146k|        fix_int_mv_precision(mv);
  472|   387k|    } else if (!hdr->hp) {
  ------------------
  |  Branch (472:16): [True: 89.6k, False: 298k]
  ------------------
  473|  89.6k|        mv->x = (mv->x - (mv->x >> 15)) & ~1U;
  474|  89.6k|        mv->y = (mv->y - (mv->y >> 15)) & ~1U;
  475|  89.6k|    }
  476|   534k|}
decode.c:fix_int_mv_precision:
  462|   167k|static inline void fix_int_mv_precision(mv *const mv) {
  463|   167k|    mv->x = (mv->x - (mv->x >> 15) + 3) & ~7U;
  464|   167k|    mv->y = (mv->y - (mv->y >> 15) + 3) & ~7U;
  465|   167k|}
decode.c:get_comp_dir_ctx:
  190|   154k|{
  191|   154k|#define has_uni_comp(edge, off) \
  192|   154k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  193|       |
  194|   154k|    if (have_top && have_left) {
  ------------------
  |  Branch (194:9): [True: 131k, False: 22.8k]
  |  Branch (194:21): [True: 125k, False: 5.48k]
  ------------------
  195|   125k|        const int a_intra = a->intra[xb4], l_intra = l->intra[yb4];
  196|       |
  197|   125k|        if (a_intra && l_intra) return 2;
  ------------------
  |  Branch (197:13): [True: 5.07k, False: 120k]
  |  Branch (197:24): [True: 1.52k, False: 3.54k]
  ------------------
  198|   124k|        if (a_intra || l_intra) {
  ------------------
  |  Branch (198:13): [True: 3.54k, False: 120k]
  |  Branch (198:24): [True: 2.61k, False: 118k]
  ------------------
  199|  6.15k|            const BlockContext *const edge = a_intra ? l : a;
  ------------------
  |  Branch (199:46): [True: 3.54k, False: 2.61k]
  ------------------
  200|  6.15k|            const int off = a_intra ? yb4 : xb4;
  ------------------
  |  Branch (200:29): [True: 3.54k, False: 2.61k]
  ------------------
  201|       |
  202|  6.15k|            if (edge->comp_type[off] == COMP_INTER_NONE) return 2;
  ------------------
  |  Branch (202:17): [True: 1.32k, False: 4.83k]
  ------------------
  203|  4.83k|            return 1 + 2 * has_uni_comp(edge, off);
  ------------------
  |  |  192|  4.83k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  204|  6.15k|        }
  205|       |
  206|   118k|        const int a_comp = a->comp_type[xb4] != COMP_INTER_NONE;
  207|   118k|        const int l_comp = l->comp_type[yb4] != COMP_INTER_NONE;
  208|   118k|        const int a_ref0 = a->ref[0][xb4], l_ref0 = l->ref[0][yb4];
  209|       |
  210|   118k|        if (!a_comp && !l_comp) {
  ------------------
  |  Branch (210:13): [True: 31.0k, False: 86.9k]
  |  Branch (210:24): [True: 9.21k, False: 21.8k]
  ------------------
  211|  9.21k|            return 1 + 2 * ((a_ref0 >= 4) == (l_ref0 >= 4));
  212|   108k|        } else if (!a_comp || !l_comp) {
  ------------------
  |  Branch (212:20): [True: 21.8k, False: 86.9k]
  |  Branch (212:31): [True: 16.1k, False: 70.8k]
  ------------------
  213|  37.9k|            const BlockContext *const edge = a_comp ? a : l;
  ------------------
  |  Branch (213:46): [True: 16.1k, False: 21.8k]
  ------------------
  214|  37.9k|            const int off = a_comp ? xb4 : yb4;
  ------------------
  |  Branch (214:29): [True: 16.1k, False: 21.8k]
  ------------------
  215|       |
  216|  37.9k|            if (!has_uni_comp(edge, off)) return 1;
  ------------------
  |  |  192|  37.9k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  |  Branch (216:17): [True: 31.1k, False: 6.80k]
  ------------------
  217|  6.80k|            return 3 + ((a_ref0 >= 4) == (l_ref0 >= 4));
  218|  70.8k|        } else {
  219|  70.8k|            const int a_uni = has_uni_comp(a, xb4), l_uni = has_uni_comp(l, yb4);
  ------------------
  |  |  192|  70.8k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
                          const int a_uni = has_uni_comp(a, xb4), l_uni = has_uni_comp(l, yb4);
  ------------------
  |  |  192|  70.8k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  220|       |
  221|  70.8k|            if (!a_uni && !l_uni) return 0;
  ------------------
  |  Branch (221:17): [True: 60.3k, False: 10.4k]
  |  Branch (221:27): [True: 55.8k, False: 4.51k]
  ------------------
  222|  15.0k|            if (!a_uni || !l_uni) return 2;
  ------------------
  |  Branch (222:17): [True: 4.51k, False: 10.4k]
  |  Branch (222:27): [True: 5.77k, False: 4.72k]
  ------------------
  223|  4.72k|            return 3 + ((a_ref0 == 4) == (l_ref0 == 4));
  224|  15.0k|        }
  225|   118k|    } else if (have_top || have_left) {
  ------------------
  |  Branch (225:16): [True: 5.48k, False: 22.8k]
  |  Branch (225:28): [True: 21.9k, False: 907]
  ------------------
  226|  27.4k|        const BlockContext *const edge = have_left ? l : a;
  ------------------
  |  Branch (226:42): [True: 21.9k, False: 5.48k]
  ------------------
  227|  27.4k|        const int off = have_left ? yb4 : xb4;
  ------------------
  |  Branch (227:25): [True: 21.9k, False: 5.48k]
  ------------------
  228|       |
  229|  27.4k|        if (edge->intra[off]) return 2;
  ------------------
  |  Branch (229:13): [True: 1.12k, False: 26.2k]
  ------------------
  230|  26.2k|        if (edge->comp_type[off] == COMP_INTER_NONE) return 2;
  ------------------
  |  Branch (230:13): [True: 6.28k, False: 20.0k]
  ------------------
  231|  20.0k|        return 4 * has_uni_comp(edge, off);
  ------------------
  |  |  192|  20.0k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  232|  26.2k|    } else {
  233|    907|        return 2;
  234|    907|    }
  235|   154k|}
decode.c:av1_get_fwd_ref_ctx:
  307|   410k|{
  308|   410k|    int cnt[4] = { 0 };
  309|       |
  310|   410k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (310:9): [True: 361k, False: 48.8k]
  |  Branch (310:21): [True: 341k, False: 20.3k]
  ------------------
  311|   341k|        if (a->ref[0][xb4] < 4) cnt[a->ref[0][xb4]]++;
  ------------------
  |  Branch (311:13): [True: 310k, False: 30.6k]
  ------------------
  312|   341k|        if (a->comp_type[xb4] && a->ref[1][xb4] < 4) cnt[a->ref[1][xb4]]++;
  ------------------
  |  Branch (312:13): [True: 102k, False: 239k]
  |  Branch (312:34): [True: 12.0k, False: 89.9k]
  ------------------
  313|   341k|    }
  314|       |
  315|   410k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (315:9): [True: 397k, False: 12.9k]
  |  Branch (315:22): [True: 377k, False: 20.3k]
  ------------------
  316|   377k|        if (l->ref[0][yb4] < 4) cnt[l->ref[0][yb4]]++;
  ------------------
  |  Branch (316:13): [True: 348k, False: 28.8k]
  ------------------
  317|   377k|        if (l->comp_type[yb4] && l->ref[1][yb4] < 4) cnt[l->ref[1][yb4]]++;
  ------------------
  |  Branch (317:13): [True: 125k, False: 251k]
  |  Branch (317:34): [True: 13.4k, False: 112k]
  ------------------
  318|   377k|    }
  319|       |
  320|   410k|    cnt[0] += cnt[1];
  321|   410k|    cnt[2] += cnt[3];
  322|       |
  323|   410k|    return cnt[0] == cnt[2] ? 1 : cnt[0] < cnt[2] ? 0 : 2;
  ------------------
  |  Branch (323:12): [True: 65.9k, False: 344k]
  |  Branch (323:35): [True: 84.9k, False: 259k]
  ------------------
  324|   410k|}
decode.c:av1_get_fwd_ref_2_ctx:
  350|   124k|{
  351|   124k|    int cnt[2] = { 0 };
  352|       |
  353|   124k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (353:9): [True: 102k, False: 22.7k]
  |  Branch (353:21): [True: 96.6k, False: 5.54k]
  ------------------
  354|  96.6k|        if ((a->ref[0][xb4] ^ 2U) < 2) cnt[a->ref[0][xb4] - 2]++;
  ------------------
  |  Branch (354:13): [True: 58.6k, False: 37.9k]
  ------------------
  355|  96.6k|        if (a->comp_type[xb4] && (a->ref[1][xb4] ^ 2U) < 2) cnt[a->ref[1][xb4] - 2]++;
  ------------------
  |  Branch (355:13): [True: 44.6k, False: 52.0k]
  |  Branch (355:34): [True: 6.02k, False: 38.6k]
  ------------------
  356|  96.6k|    }
  357|       |
  358|   124k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (358:9): [True: 119k, False: 5.55k]
  |  Branch (358:22): [True: 114k, False: 5.26k]
  ------------------
  359|   114k|        if ((l->ref[0][yb4] ^ 2U) < 2) cnt[l->ref[0][yb4] - 2]++;
  ------------------
  |  Branch (359:13): [True: 78.0k, False: 36.1k]
  ------------------
  360|   114k|        if (l->comp_type[yb4] && (l->ref[1][yb4] ^ 2U) < 2) cnt[l->ref[1][yb4] - 2]++;
  ------------------
  |  Branch (360:13): [True: 62.9k, False: 51.2k]
  |  Branch (360:34): [True: 7.16k, False: 55.7k]
  ------------------
  361|   114k|    }
  362|       |
  363|   124k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (363:12): [True: 29.2k, False: 95.6k]
  |  Branch (363:35): [True: 72.6k, False: 23.0k]
  ------------------
  364|   124k|}
decode.c:av1_get_fwd_ref_1_ctx:
  330|   297k|{
  331|   297k|    int cnt[2] = { 0 };
  332|       |
  333|   297k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (333:9): [True: 270k, False: 27.2k]
  |  Branch (333:21): [True: 255k, False: 15.1k]
  ------------------
  334|   255k|        if (a->ref[0][xb4] < 2) cnt[a->ref[0][xb4]]++;
  ------------------
  |  Branch (334:13): [True: 220k, False: 34.8k]
  ------------------
  335|   255k|        if (a->comp_type[xb4] && a->ref[1][xb4] < 2) cnt[a->ref[1][xb4]]++;
  ------------------
  |  Branch (335:13): [True: 64.5k, False: 190k]
  |  Branch (335:34): [True: 4.02k, False: 60.5k]
  ------------------
  336|   255k|    }
  337|       |
  338|   297k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (338:9): [True: 288k, False: 9.19k]
  |  Branch (338:22): [True: 273k, False: 15.2k]
  ------------------
  339|   273k|        if (l->ref[0][yb4] < 2) cnt[l->ref[0][yb4]]++;
  ------------------
  |  Branch (339:13): [True: 236k, False: 36.2k]
  ------------------
  340|   273k|        if (l->comp_type[yb4] && l->ref[1][yb4] < 2) cnt[l->ref[1][yb4]]++;
  ------------------
  |  Branch (340:13): [True: 70.0k, False: 202k]
  |  Branch (340:34): [True: 4.58k, False: 65.4k]
  ------------------
  341|   273k|    }
  342|       |
  343|   297k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (343:12): [True: 40.6k, False: 256k]
  |  Branch (343:35): [True: 18.4k, False: 238k]
  ------------------
  344|   297k|}
decode.c:av1_get_bwd_ref_ctx:
  370|   335k|{
  371|   335k|    int cnt[3] = { 0 };
  372|       |
  373|   335k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (373:9): [True: 275k, False: 60.7k]
  |  Branch (373:21): [True: 261k, False: 13.4k]
  ------------------
  374|   261k|        if (a->ref[0][xb4] >= 4) cnt[a->ref[0][xb4] - 4]++;
  ------------------
  |  Branch (374:13): [True: 137k, False: 123k]
  ------------------
  375|   261k|        if (a->comp_type[xb4] && a->ref[1][xb4] >= 4) cnt[a->ref[1][xb4] - 4]++;
  ------------------
  |  Branch (375:13): [True: 91.7k, False: 169k]
  |  Branch (375:34): [True: 85.6k, False: 6.17k]
  ------------------
  376|   261k|    }
  377|       |
  378|   335k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (378:9): [True: 323k, False: 12.7k]
  |  Branch (378:22): [True: 309k, False: 14.1k]
  ------------------
  379|   309k|        if (l->ref[0][yb4] >= 4) cnt[l->ref[0][yb4] - 4]++;
  ------------------
  |  Branch (379:13): [True: 165k, False: 143k]
  ------------------
  380|   309k|        if (l->comp_type[yb4] && l->ref[1][yb4] >= 4) cnt[l->ref[1][yb4] - 4]++;
  ------------------
  |  Branch (380:13): [True: 113k, False: 195k]
  |  Branch (380:34): [True: 107k, False: 5.73k]
  ------------------
  381|   309k|    }
  382|       |
  383|   335k|    cnt[1] += cnt[0];
  384|       |
  385|   335k|    return cnt[2] == cnt[1] ? 1 : cnt[1] < cnt[2] ? 0 : 2;
  ------------------
  |  Branch (385:12): [True: 51.5k, False: 284k]
  |  Branch (385:35): [True: 194k, False: 89.6k]
  ------------------
  386|   335k|}
decode.c:av1_get_bwd_ref_1_ctx:
  392|   113k|{
  393|   113k|    int cnt[3] = { 0 };
  394|       |
  395|   113k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (395:9): [True: 102k, False: 11.5k]
  |  Branch (395:21): [True: 95.7k, False: 6.38k]
  ------------------
  396|  95.7k|        if (a->ref[0][xb4] >= 4) cnt[a->ref[0][xb4] - 4]++;
  ------------------
  |  Branch (396:13): [True: 34.7k, False: 60.9k]
  ------------------
  397|  95.7k|        if (a->comp_type[xb4] && a->ref[1][xb4] >= 4) cnt[a->ref[1][xb4] - 4]++;
  ------------------
  |  Branch (397:13): [True: 46.1k, False: 49.6k]
  |  Branch (397:34): [True: 42.3k, False: 3.73k]
  ------------------
  398|  95.7k|    }
  399|       |
  400|   113k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (400:9): [True: 108k, False: 4.93k]
  |  Branch (400:22): [True: 101k, False: 7.11k]
  ------------------
  401|   101k|        if (l->ref[0][yb4] >= 4) cnt[l->ref[0][yb4] - 4]++;
  ------------------
  |  Branch (401:13): [True: 34.7k, False: 66.9k]
  ------------------
  402|   101k|        if (l->comp_type[yb4] && l->ref[1][yb4] >= 4) cnt[l->ref[1][yb4] - 4]++;
  ------------------
  |  Branch (402:13): [True: 52.6k, False: 49.0k]
  |  Branch (402:34): [True: 49.5k, False: 3.12k]
  ------------------
  403|   101k|    }
  404|       |
  405|   113k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (405:12): [True: 24.3k, False: 89.3k]
  |  Branch (405:35): [True: 34.5k, False: 54.8k]
  ------------------
  406|   113k|}
decode.c:av1_get_ref_ctx:
  287|   508k|{
  288|   508k|    int cnt[2] = { 0 };
  289|       |
  290|   508k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (290:9): [True: 437k, False: 71.4k]
  |  Branch (290:21): [True: 410k, False: 26.5k]
  ------------------
  291|   410k|        cnt[a->ref[0][xb4] >= 4]++;
  292|   410k|        if (a->comp_type[xb4]) cnt[a->ref[1][xb4] >= 4]++;
  ------------------
  |  Branch (292:13): [True: 48.0k, False: 362k]
  ------------------
  293|   410k|    }
  294|       |
  295|   508k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (295:9): [True: 489k, False: 19.3k]
  |  Branch (295:22): [True: 463k, False: 26.2k]
  ------------------
  296|   463k|        cnt[l->ref[0][yb4] >= 4]++;
  297|   463k|        if (l->comp_type[yb4]) cnt[l->ref[1][yb4] >= 4]++;
  ------------------
  |  Branch (297:13): [True: 58.9k, False: 404k]
  ------------------
  298|   463k|    }
  299|       |
  300|   508k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (300:12): [True: 71.8k, False: 436k]
  |  Branch (300:35): [True: 176k, False: 260k]
  ------------------
  301|   508k|}
decode.c:av1_get_uni_p1_ctx:
  412|  18.6k|{
  413|  18.6k|    int cnt[3] = { 0 };
  414|       |
  415|  18.6k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (415:9): [True: 16.5k, False: 2.02k]
  |  Branch (415:21): [True: 15.8k, False: 782]
  ------------------
  416|  15.8k|        if (a->ref[0][xb4] - 1U < 3) cnt[a->ref[0][xb4] - 1]++;
  ------------------
  |  Branch (416:13): [True: 3.96k, False: 11.8k]
  ------------------
  417|  15.8k|        if (a->comp_type[xb4] && a->ref[1][xb4] - 1U < 3) cnt[a->ref[1][xb4] - 1]++;
  ------------------
  |  Branch (417:13): [True: 10.7k, False: 5.09k]
  |  Branch (417:34): [True: 6.84k, False: 3.87k]
  ------------------
  418|  15.8k|    }
  419|       |
  420|  18.6k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (420:9): [True: 16.6k, False: 1.96k]
  |  Branch (420:22): [True: 16.0k, False: 627]
  ------------------
  421|  16.0k|        if (l->ref[0][yb4] - 1U < 3) cnt[l->ref[0][yb4] - 1]++;
  ------------------
  |  Branch (421:13): [True: 4.14k, False: 11.8k]
  ------------------
  422|  16.0k|        if (l->comp_type[yb4] && l->ref[1][yb4] - 1U < 3) cnt[l->ref[1][yb4] - 1]++;
  ------------------
  |  Branch (422:13): [True: 11.5k, False: 4.45k]
  |  Branch (422:34): [True: 6.50k, False: 5.07k]
  ------------------
  423|  16.0k|    }
  424|       |
  425|  18.6k|    cnt[1] += cnt[2];
  426|       |
  427|  18.6k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (427:12): [True: 5.26k, False: 13.3k]
  |  Branch (427:35): [True: 8.99k, False: 4.36k]
  ------------------
  428|  18.6k|}
decode.c:get_drl_context:
  432|   296k|{
  433|   296k|    if (ref_mv_stack[ref_idx].weight >= 640)
  ------------------
  |  Branch (433:9): [True: 240k, False: 55.8k]
  ------------------
  434|   240k|        return ref_mv_stack[ref_idx + 1].weight < 640;
  435|       |
  436|  55.8k|    return ref_mv_stack[ref_idx + 1].weight < 640 ? 2 : 0;
  ------------------
  |  Branch (436:12): [True: 55.8k, False: 0]
  ------------------
  437|   296k|}
decode.c:get_gmv_2d:
  482|   437k|{
  483|   437k|    switch (gmv->type) {
  484|  59.8k|    case DAV1D_WM_TYPE_ROT_ZOOM:
  ------------------
  |  Branch (484:5): [True: 59.8k, False: 377k]
  ------------------
  485|  59.8k|        assert(gmv->matrix[5] ==  gmv->matrix[2]);
  ------------------
  |  Branch (485:9): [True: 59.8k, False: 0]
  ------------------
  486|  59.8k|        assert(gmv->matrix[4] == -gmv->matrix[3]);
  ------------------
  |  Branch (486:9): [True: 59.8k, False: 0]
  ------------------
  487|       |        // fall-through
  488|  59.8k|    default:
  ------------------
  |  Branch (488:5): [True: 0, False: 437k]
  ------------------
  489|  65.9k|    case DAV1D_WM_TYPE_AFFINE: {
  ------------------
  |  Branch (489:5): [True: 6.16k, False: 430k]
  ------------------
  490|  65.9k|        const int x = bx4 * 4 + bw4 * 2 - 1;
  491|  65.9k|        const int y = by4 * 4 + bh4 * 2 - 1;
  492|  65.9k|        const int xc = (gmv->matrix[2] - (1 << 16)) * x +
  493|  65.9k|                       gmv->matrix[3] * y + gmv->matrix[0];
  494|  65.9k|        const int yc = (gmv->matrix[5] - (1 << 16)) * y +
  495|  65.9k|                       gmv->matrix[4] * x + gmv->matrix[1];
  496|  65.9k|        const int shift = 16 - (3 - !hdr->hp);
  497|  65.9k|        const int round = (1 << shift) >> 1;
  498|  65.9k|        mv res = (mv) {
  499|  65.9k|            .y = apply_sign(((abs(yc) + round) >> shift) << !hdr->hp, yc),
  500|  65.9k|            .x = apply_sign(((abs(xc) + round) >> shift) << !hdr->hp, xc),
  501|  65.9k|        };
  502|  65.9k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (502:13): [True: 16.0k, False: 49.9k]
  ------------------
  503|  16.0k|            fix_int_mv_precision(&res);
  504|  65.9k|        return res;
  505|  59.8k|    }
  506|   144k|    case DAV1D_WM_TYPE_TRANSLATION: {
  ------------------
  |  Branch (506:5): [True: 144k, False: 293k]
  ------------------
  507|   144k|        mv res = (mv) {
  508|   144k|            .y = gmv->matrix[0] >> 13,
  509|   144k|            .x = gmv->matrix[1] >> 13,
  510|   144k|        };
  511|   144k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (511:13): [True: 5.41k, False: 138k]
  ------------------
  512|  5.41k|            fix_int_mv_precision(&res);
  513|   144k|        return res;
  514|  59.8k|    }
  515|   227k|    case DAV1D_WM_TYPE_IDENTITY:
  ------------------
  |  Branch (515:5): [True: 227k, False: 210k]
  ------------------
  516|   227k|        return (mv) { .x = 0, .y = 0 };
  517|   437k|    }
  518|   437k|}
decode.c:get_mask_comp_ctx:
  266|   142k|{
  267|   142k|    const int a_ctx = a->comp_type[xb4] >= COMP_INTER_SEG ? 1 :
  ------------------
  |  Branch (267:23): [True: 25.9k, False: 116k]
  ------------------
  268|   142k|                      a->ref[0][xb4] == 6 ? 3 : 0;
  ------------------
  |  Branch (268:23): [True: 6.76k, False: 109k]
  ------------------
  269|   142k|    const int l_ctx = l->comp_type[yb4] >= COMP_INTER_SEG ? 1 :
  ------------------
  |  Branch (269:23): [True: 37.9k, False: 104k]
  ------------------
  270|   142k|                      l->ref[0][yb4] == 6 ? 3 : 0;
  ------------------
  |  Branch (270:23): [True: 6.08k, False: 97.9k]
  ------------------
  271|       |
  272|   142k|    return imin(a_ctx + l_ctx, 5);
  273|   142k|}
decode.c:get_jnt_comp_ctx:
  251|  75.5k|{
  252|  75.5k|    const int d0 = abs(get_poc_diff(order_hint_n_bits, ref0poc, poc));
  253|  75.5k|    const int d1 = abs(get_poc_diff(order_hint_n_bits, poc, ref1poc));
  254|  75.5k|    const int offset = d0 == d1;
  255|  75.5k|    const int a_ctx = a->comp_type[xb4] >= COMP_INTER_AVG ||
  ------------------
  |  Branch (255:23): [True: 36.8k, False: 38.6k]
  ------------------
  256|  38.6k|                      a->ref[0][xb4] == 6;
  ------------------
  |  Branch (256:23): [True: 3.53k, False: 35.1k]
  ------------------
  257|  75.5k|    const int l_ctx = l->comp_type[yb4] >= COMP_INTER_AVG ||
  ------------------
  |  Branch (257:23): [True: 38.0k, False: 37.5k]
  ------------------
  258|  37.5k|                      l->ref[0][yb4] == 6;
  ------------------
  |  Branch (258:23): [True: 2.95k, False: 34.5k]
  ------------------
  259|       |
  260|  75.5k|    return 3 * offset + a_ctx + l_ctx;
  261|  75.5k|}
decode.c:get_filter_ctx:
  139|   581k|{
  140|   581k|    const int a_filter = (a->ref[0][xb4] == ref || a->ref[1][xb4] == ref) ?
  ------------------
  |  Branch (140:27): [True: 390k, False: 190k]
  |  Branch (140:52): [True: 11.8k, False: 178k]
  ------------------
  141|   402k|                         a->filter[dir][xb4] : DAV1D_N_SWITCHABLE_FILTERS;
  142|   581k|    const int l_filter = (l->ref[0][yb4] == ref || l->ref[1][yb4] == ref) ?
  ------------------
  |  Branch (142:27): [True: 445k, False: 135k]
  |  Branch (142:52): [True: 12.9k, False: 122k]
  ------------------
  143|   458k|                         l->filter[dir][yb4] : DAV1D_N_SWITCHABLE_FILTERS;
  144|       |
  145|   581k|    if (a_filter == l_filter) {
  ------------------
  |  Branch (145:9): [True: 368k, False: 212k]
  ------------------
  146|   368k|        return comp * 4 + a_filter;
  147|   368k|    } else if (a_filter == DAV1D_N_SWITCHABLE_FILTERS) {
  ------------------
  |  Branch (147:16): [True: 120k, False: 92.0k]
  ------------------
  148|   120k|        return comp * 4 + l_filter;
  149|   120k|    } else if (l_filter == DAV1D_N_SWITCHABLE_FILTERS) {
  ------------------
  |  Branch (149:16): [True: 64.4k, False: 27.6k]
  ------------------
  150|  64.4k|        return comp * 4 + a_filter;
  151|  64.4k|    } else {
  152|  27.6k|        return comp * 4 + DAV1D_N_SWITCHABLE_FILTERS;
  153|  27.6k|    }
  154|   581k|}
decode.c:gather_top_partition_prob:
  106|   526k|{
  107|       |    // Exploit the fact that cdfs for PARTITION_V, PARTITION_SPLIT and
  108|       |    // PARTITION_T_TOP_SPLIT are neighbors.
  109|   526k|    unsigned out = in[PARTITION_V - 1] - in[PARTITION_T_TOP_SPLIT];
  110|       |    // Exploit the facts that cdfs for PARTITION_T_LEFT_SPLIT and
  111|       |    // PARTITION_T_RIGHT_SPLIT are neighbors, the probability for
  112|       |    // PARTITION_V4 is always zero, and the probability for
  113|       |    // PARTITION_T_RIGHT_SPLIT is zero in 128x128 blocks.
  114|   526k|    out += in[PARTITION_T_LEFT_SPLIT - 1];
  115|   526k|    if (bl != BL_128X128)
  ------------------
  |  Branch (115:9): [True: 475k, False: 50.4k]
  ------------------
  116|   475k|        out += in[PARTITION_V4 - 1] - in[PARTITION_T_RIGHT_SPLIT];
  117|   526k|    return out;
  118|   526k|}
decode.c:gather_left_partition_prob:
   94|  62.6k|{
   95|  62.6k|    unsigned out = in[PARTITION_H - 1] - in[PARTITION_H];
   96|       |    // Exploit the fact that cdfs for PARTITION_SPLIT, PARTITION_T_TOP_SPLIT,
   97|       |    // PARTITION_T_BOTTOM_SPLIT and PARTITION_T_LEFT_SPLIT are neighbors.
   98|  62.6k|    out += in[PARTITION_SPLIT - 1] - in[PARTITION_T_LEFT_SPLIT];
   99|  62.6k|    if (bl != BL_128X128)
  ------------------
  |  Branch (99:9): [True: 51.7k, False: 10.9k]
  ------------------
  100|  51.7k|        out += in[PARTITION_H4 - 1] - in[PARTITION_H4];
  101|  62.6k|    return out;
  102|  62.6k|}
decode.c:get_poc_diff:
  239|   581k|{
  240|   581k|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 40.9k, False: 540k]
  ------------------
  241|   540k|    const int mask = 1 << (order_hint_n_bits - 1);
  242|   540k|    const int diff = poc0 - poc1;
  243|   540k|    return (diff & (mask - 1)) - (diff & mask);
  244|   581k|}
recon_tmpl.c:get_uv_inter_txtp:
  122|   197k|{
  123|   197k|    if (uvt_dim->max == TX_32X32)
  ------------------
  |  Branch (123:9): [True: 49.5k, False: 148k]
  ------------------
  124|  49.5k|        return ytxtp == IDTX ? IDTX : DCT_DCT;
  ------------------
  |  Branch (124:16): [True: 2.38k, False: 47.1k]
  ------------------
  125|   148k|    if (uvt_dim->min == TX_16X16 &&
  ------------------
  |  Branch (125:9): [True: 17.9k, False: 130k]
  ------------------
  126|  17.9k|        ((1 << ytxtp) & ((1 << H_FLIPADST) | (1 << V_FLIPADST) |
  ------------------
  |  Branch (126:9): [True: 541, False: 17.4k]
  ------------------
  127|  17.9k|                         (1 << H_ADST) | (1 << V_ADST))))
  128|    541|    {
  129|    541|        return DCT_DCT;
  130|    541|    }
  131|       |
  132|   147k|    return ytxtp;
  133|   148k|}

dav1d_prep_grain_8bpc:
  105|  1.79k|{
  106|  1.79k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  107|       |#if BITDEPTH != 8
  108|       |    const int bitdepth_max = (1 << out->p.bpc) - 1;
  109|       |#endif
  110|       |
  111|       |    // Generate grain LUTs as needed
  112|  1.79k|    dsp->generate_grain_y(grain_lut[0], data HIGHBD_TAIL_SUFFIX); // always needed
  113|  1.79k|    if (data->num_uv_points[0] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (113:9): [True: 772, False: 1.02k]
  |  Branch (113:35): [True: 241, False: 784]
  ------------------
  114|  1.01k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[1], grain_lut[0],
  115|  1.01k|                                                 data, 0 HIGHBD_TAIL_SUFFIX);
  116|  1.79k|    if (data->num_uv_points[1] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (116:9): [True: 821, False: 976]
  |  Branch (116:35): [True: 241, False: 735]
  ------------------
  117|  1.06k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[2], grain_lut[0],
  118|  1.06k|                                                 data, 1 HIGHBD_TAIL_SUFFIX);
  119|       |
  120|       |    // Generate scaling LUTs as needed
  121|  1.79k|    if (data->num_y_points || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (121:9): [True: 1.17k, False: 627]
  |  Branch (121:31): [True: 206, False: 421]
  ------------------
  122|  1.37k|        generate_scaling(in->p.bpc, data->y_points, data->num_y_points, scaling[0]);
  123|  1.79k|    if (data->num_uv_points[0])
  ------------------
  |  Branch (123:9): [True: 772, False: 1.02k]
  ------------------
  124|    772|        generate_scaling(in->p.bpc, data->uv_points[0], data->num_uv_points[0], scaling[1]);
  125|  1.79k|    if (data->num_uv_points[1])
  ------------------
  |  Branch (125:9): [True: 821, False: 976]
  ------------------
  126|    821|        generate_scaling(in->p.bpc, data->uv_points[1], data->num_uv_points[1], scaling[2]);
  127|       |
  128|       |    // Copy over the non-modified planes
  129|  1.79k|    assert(out->stride[0] == in->stride[0]);
  ------------------
  |  Branch (129:5): [True: 1.79k, False: 0]
  ------------------
  130|  1.79k|    if (!data->num_y_points) {
  ------------------
  |  Branch (130:9): [True: 627, False: 1.17k]
  ------------------
  131|    627|        const ptrdiff_t stride = out->stride[0];
  132|    627|        const ptrdiff_t sz = out->p.h * stride;
  133|    627|        if (sz < 0)
  ------------------
  |  Branch (133:13): [True: 0, False: 627]
  ------------------
  134|      0|            memcpy((uint8_t*) out->data[0] + sz - stride,
  135|      0|                   (uint8_t*) in->data[0] + sz - stride, -sz);
  136|    627|        else
  137|    627|            memcpy(out->data[0], in->data[0], sz);
  138|    627|    }
  139|       |
  140|  1.79k|    if (in->p.layout != DAV1D_PIXEL_LAYOUT_I400 && !data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (140:9): [True: 1.44k, False: 355]
  |  Branch (140:52): [True: 1.20k, False: 241]
  ------------------
  141|  1.20k|        assert(out->stride[1] == in->stride[1]);
  ------------------
  |  Branch (141:9): [True: 1.20k, False: 0]
  ------------------
  142|  1.20k|        const int ss_ver = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  143|  1.20k|        const ptrdiff_t stride = out->stride[1];
  144|  1.20k|        const ptrdiff_t sz = ((out->p.h + ss_ver) >> ss_ver) * stride;
  145|  1.20k|        if (sz < 0) {
  ------------------
  |  Branch (145:13): [True: 0, False: 1.20k]
  ------------------
  146|      0|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (146:17): [True: 0, False: 0]
  ------------------
  147|      0|                memcpy((uint8_t*) out->data[1] + sz - stride,
  148|      0|                       (uint8_t*) in->data[1] + sz - stride, -sz);
  149|      0|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (149:17): [True: 0, False: 0]
  ------------------
  150|      0|                memcpy((uint8_t*) out->data[2] + sz - stride,
  151|      0|                       (uint8_t*) in->data[2] + sz - stride, -sz);
  152|  1.20k|        } else {
  153|  1.20k|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (153:17): [True: 429, False: 772]
  ------------------
  154|    429|                memcpy(out->data[1], in->data[1], sz);
  155|  1.20k|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (155:17): [True: 380, False: 821]
  ------------------
  156|    380|                memcpy(out->data[2], in->data[2], sz);
  157|  1.20k|        }
  158|  1.20k|    }
  159|  1.79k|}
dav1d_apply_grain_row_8bpc:
  167|  14.3k|{
  168|       |    // Synthesize grain for the affected planes
  169|  14.3k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  170|  14.3k|    const int ss_y = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  171|  14.3k|    const int ss_x = in->p.layout != DAV1D_PIXEL_LAYOUT_I444;
  172|  14.3k|    const int cpw = (out->p.w + ss_x) >> ss_x;
  173|  14.3k|    const int is_id = out->seq_hdr->mtrx == DAV1D_MC_IDENTITY;
  174|  14.3k|    pixel *const luma_src =
  175|  14.3k|        ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   37|  14.3k|#define FG_BLOCK_SIZE 32
  ------------------
                      ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   53|  14.3k|#define PXSTRIDE(x) (x)
  ------------------
  176|       |#if BITDEPTH != 8
  177|       |    const int bitdepth_max = (1 << out->p.bpc) - 1;
  178|       |#endif
  179|       |
  180|  14.3k|    if (data->num_y_points) {
  ------------------
  |  Branch (180:9): [True: 10.6k, False: 3.74k]
  ------------------
  181|  10.6k|        const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  10.6k|#define FG_BLOCK_SIZE 32
  ------------------
                      const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  10.6k|#define FG_BLOCK_SIZE 32
  ------------------
  182|  10.6k|        dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   37|  10.6k|#define FG_BLOCK_SIZE 32
  ------------------
                      dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   53|  10.6k|#define PXSTRIDE(x) (x)
  ------------------
  183|  10.6k|                         luma_src, out->stride[0], data,
  184|  10.6k|                         out->p.w, scaling[0], grain_lut[0], bh, row HIGHBD_TAIL_SUFFIX);
  185|  10.6k|    }
  186|       |
  187|  14.3k|    if (!data->num_uv_points[0] && !data->num_uv_points[1] &&
  ------------------
  |  Branch (187:9): [True: 11.4k, False: 2.91k]
  |  Branch (187:36): [True: 10.3k, False: 1.13k]
  ------------------
  188|  10.3k|        !data->chroma_scaling_from_luma)
  ------------------
  |  Branch (188:9): [True: 2.62k, False: 7.71k]
  ------------------
  189|  2.62k|    {
  190|  2.62k|        return;
  191|  2.62k|    }
  192|       |
  193|  11.7k|    const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  11.7k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  11.7k|#define FG_BLOCK_SIZE 32
  ------------------
  194|       |
  195|       |    // extend padding pixels
  196|  11.7k|    if (out->p.w & ss_x) {
  ------------------
  |  Branch (196:9): [True: 4.26k, False: 7.50k]
  ------------------
  197|  4.26k|        pixel *ptr = luma_src;
  198|   132k|        for (int y = 0; y < bh; y++) {
  ------------------
  |  Branch (198:25): [True: 128k, False: 4.26k]
  ------------------
  199|   128k|            ptr[out->p.w] = ptr[out->p.w - 1];
  200|   128k|            ptr += PXSTRIDE(in->stride[0]) << ss_y;
  ------------------
  |  |   53|   128k|#define PXSTRIDE(x) (x)
  ------------------
  201|   128k|        }
  202|  4.26k|    }
  203|       |
  204|  11.7k|    const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   37|  11.7k|#define FG_BLOCK_SIZE 32
  ------------------
                  const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   53|  11.7k|#define PXSTRIDE(x) (x)
  ------------------
  205|  11.7k|    if (data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (205:9): [True: 7.71k, False: 4.04k]
  ------------------
  206|  23.1k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (206:26): [True: 15.4k, False: 7.71k]
  ------------------
  207|  15.4k|            dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  208|  15.4k|                                                ((const pixel *) in->data[1 + pl]) + uv_off,
  209|  15.4k|                                                in->stride[1], data, cpw,
  210|  15.4k|                                                scaling[0], grain_lut[1 + pl],
  211|  15.4k|                                                bh, row, luma_src, in->stride[0],
  212|  15.4k|                                                pl, is_id HIGHBD_TAIL_SUFFIX);
  213|  7.71k|    } else {
  214|  12.1k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (214:26): [True: 8.09k, False: 4.04k]
  ------------------
  215|  8.09k|            if (data->num_uv_points[pl])
  ------------------
  |  Branch (215:17): [True: 4.64k, False: 3.45k]
  ------------------
  216|  4.64k|                dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  217|  4.64k|                                                    ((const pixel *) in->data[1 + pl]) + uv_off,
  218|  4.64k|                                                    in->stride[1], data, cpw,
  219|  4.64k|                                                    scaling[1 + pl], grain_lut[1 + pl],
  220|  4.64k|                                                    bh, row, luma_src, in->stride[0],
  221|  4.64k|                                                    pl, is_id HIGHBD_TAIL_SUFFIX);
  222|  4.04k|    }
  223|  11.7k|}
dav1d_apply_grain_8bpc:
  228|  1.79k|{
  229|  1.79k|    ALIGN_STK_16(entry, grain_lut, 3,[GRAIN_HEIGHT + 1][GRAIN_WIDTH]);
  ------------------
  |  |  100|  1.79k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  1.79k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  230|  1.79k|#if ARCH_X86_64 && BITDEPTH == 8
  231|  1.79k|    ALIGN_STK_64(uint8_t, scaling, 3,[SCALING_SIZE]);
  ------------------
  |  |   96|  1.79k|    ALIGN(type var[sz1d]sznd, ALIGN_64_VAL)
  |  |  ------------------
  |  |  |  |   86|  1.79k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  232|       |#else
  233|       |    uint8_t scaling[3][SCALING_SIZE];
  234|       |#endif
  235|  1.79k|    const int rows = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  1.79k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int rows = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  1.79k|#define FG_BLOCK_SIZE 32
  ------------------
  236|       |
  237|  1.79k|    bitfn(dav1d_prep_grain)(dsp, out, in, scaling, grain_lut);
  ------------------
  |  |   51|  1.79k|#define bitfn(x) x##_8bpc
  ------------------
  238|  16.1k|    for (int row = 0; row < rows; row++)
  ------------------
  |  Branch (238:23): [True: 14.3k, False: 1.79k]
  ------------------
  239|  14.3k|        bitfn(dav1d_apply_grain_row)(dsp, out, in, scaling, grain_lut, row);
  ------------------
  |  |   51|  14.3k|#define bitfn(x) x##_8bpc
  ------------------
  240|  1.79k|}
fg_apply_tmpl.c:generate_scaling:
   44|  2.96k|{
   45|  2.96k|#if BITDEPTH == 8
   46|  2.96k|    const int shift_x = 0;
   47|  2.96k|    const int scaling_size = SCALING_SIZE;
  ------------------
  |  |   39|  2.96k|#define SCALING_SIZE 256
  ------------------
   48|       |#else
   49|       |    assert(bitdepth > 8);
   50|       |    const int shift_x = bitdepth - 8;
   51|       |    const int scaling_size = 1 << bitdepth;
   52|       |#endif
   53|       |
   54|  2.96k|    if (num == 0) {
  ------------------
  |  Branch (54:9): [True: 206, False: 2.76k]
  ------------------
   55|    206|        memset(scaling, 0, scaling_size);
   56|    206|        return;
   57|    206|    }
   58|       |
   59|       |    // Fill up the preceding entries with the initial value
   60|  2.76k|    memset(scaling, points[0][1], points[0][0] << shift_x);
   61|       |
   62|       |    // Linearly interpolate the values in the middle
   63|  4.99k|    for (int i = 0; i < num - 1; i++) {
  ------------------
  |  Branch (63:21): [True: 2.23k, False: 2.76k]
  ------------------
   64|  2.23k|        const int bx = points[i][0];
   65|  2.23k|        const int by = points[i][1];
   66|  2.23k|        const int ex = points[i+1][0];
   67|  2.23k|        const int ey = points[i+1][1];
   68|  2.23k|        const int dx = ex - bx;
   69|  2.23k|        const int dy = ey - by;
   70|  2.23k|        assert(dx > 0);
  ------------------
  |  Branch (70:9): [True: 2.23k, False: 0]
  ------------------
   71|  2.23k|        const int delta = dy * ((0x10000 + (dx >> 1)) / dx);
   72|   154k|        for (int x = 0, d = 0x8000; x < dx; x++) {
  ------------------
  |  Branch (72:37): [True: 152k, False: 2.23k]
  ------------------
   73|   152k|            scaling[(bx + x) << shift_x] = by + (d >> 16);
   74|   152k|            d += delta;
   75|   152k|        }
   76|  2.23k|    }
   77|       |
   78|       |    // Fill up the remaining entries with the final value
   79|  2.76k|    const int n = points[num - 1][0] << shift_x;
   80|  2.76k|    memset(&scaling[n], points[num - 1][1], scaling_size - n);
   81|       |
   82|       |#if BITDEPTH != 8
   83|       |    const int pad = 1 << shift_x, rnd = pad >> 1;
   84|       |    for (int i = 0; i < num - 1; i++) {
   85|       |        const int bx = points[i][0] << shift_x;
   86|       |        const int ex = points[i+1][0] << shift_x;
   87|       |        const int dx = ex - bx;
   88|       |        for (int x = 0; x < dx; x += pad) {
   89|       |            const int range = scaling[bx + x + pad] - scaling[bx + x];
   90|       |            for (int n = 1, r = rnd; n < pad; n++) {
   91|       |                r += range;
   92|       |                scaling[bx + x + n] = scaling[bx + x] + (r >> shift_x);
   93|       |            }
   94|       |        }
   95|       |    }
   96|       |#endif
   97|  2.76k|}
dav1d_prep_grain_16bpc:
  105|  2.51k|{
  106|  2.51k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  107|  2.51k|#if BITDEPTH != 8
  108|  2.51k|    const int bitdepth_max = (1 << out->p.bpc) - 1;
  109|  2.51k|#endif
  110|       |
  111|       |    // Generate grain LUTs as needed
  112|  2.51k|    dsp->generate_grain_y(grain_lut[0], data HIGHBD_TAIL_SUFFIX); // always needed
  ------------------
  |  |   74|  2.51k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  113|  2.51k|    if (data->num_uv_points[0] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (113:9): [True: 472, False: 2.04k]
  |  Branch (113:35): [True: 530, False: 1.51k]
  ------------------
  114|  1.00k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[1], grain_lut[0],
  115|  1.00k|                                                 data, 0 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  1.00k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  116|  2.51k|    if (data->num_uv_points[1] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (116:9): [True: 639, False: 1.87k]
  |  Branch (116:35): [True: 530, False: 1.34k]
  ------------------
  117|  1.16k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[2], grain_lut[0],
  118|  1.16k|                                                 data, 1 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  1.16k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  119|       |
  120|       |    // Generate scaling LUTs as needed
  121|  2.51k|    if (data->num_y_points || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (121:9): [True: 1.72k, False: 792]
  |  Branch (121:31): [True: 232, False: 560]
  ------------------
  122|  1.95k|        generate_scaling(in->p.bpc, data->y_points, data->num_y_points, scaling[0]);
  123|  2.51k|    if (data->num_uv_points[0])
  ------------------
  |  Branch (123:9): [True: 472, False: 2.04k]
  ------------------
  124|    472|        generate_scaling(in->p.bpc, data->uv_points[0], data->num_uv_points[0], scaling[1]);
  125|  2.51k|    if (data->num_uv_points[1])
  ------------------
  |  Branch (125:9): [True: 639, False: 1.87k]
  ------------------
  126|    639|        generate_scaling(in->p.bpc, data->uv_points[1], data->num_uv_points[1], scaling[2]);
  127|       |
  128|       |    // Copy over the non-modified planes
  129|  2.51k|    assert(out->stride[0] == in->stride[0]);
  ------------------
  |  Branch (129:5): [True: 2.51k, False: 0]
  ------------------
  130|  2.51k|    if (!data->num_y_points) {
  ------------------
  |  Branch (130:9): [True: 792, False: 1.72k]
  ------------------
  131|    792|        const ptrdiff_t stride = out->stride[0];
  132|    792|        const ptrdiff_t sz = out->p.h * stride;
  133|    792|        if (sz < 0)
  ------------------
  |  Branch (133:13): [True: 0, False: 792]
  ------------------
  134|      0|            memcpy((uint8_t*) out->data[0] + sz - stride,
  135|      0|                   (uint8_t*) in->data[0] + sz - stride, -sz);
  136|    792|        else
  137|    792|            memcpy(out->data[0], in->data[0], sz);
  138|    792|    }
  139|       |
  140|  2.51k|    if (in->p.layout != DAV1D_PIXEL_LAYOUT_I400 && !data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (140:9): [True: 1.57k, False: 942]
  |  Branch (140:52): [True: 1.04k, False: 530]
  ------------------
  141|  1.04k|        assert(out->stride[1] == in->stride[1]);
  ------------------
  |  Branch (141:9): [True: 1.04k, False: 0]
  ------------------
  142|  1.04k|        const int ss_ver = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  143|  1.04k|        const ptrdiff_t stride = out->stride[1];
  144|  1.04k|        const ptrdiff_t sz = ((out->p.h + ss_ver) >> ss_ver) * stride;
  145|  1.04k|        if (sz < 0) {
  ------------------
  |  Branch (145:13): [True: 0, False: 1.04k]
  ------------------
  146|      0|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (146:17): [True: 0, False: 0]
  ------------------
  147|      0|                memcpy((uint8_t*) out->data[1] + sz - stride,
  148|      0|                       (uint8_t*) in->data[1] + sz - stride, -sz);
  149|      0|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (149:17): [True: 0, False: 0]
  ------------------
  150|      0|                memcpy((uint8_t*) out->data[2] + sz - stride,
  151|      0|                       (uint8_t*) in->data[2] + sz - stride, -sz);
  152|  1.04k|        } else {
  153|  1.04k|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (153:17): [True: 569, False: 472]
  ------------------
  154|    569|                memcpy(out->data[1], in->data[1], sz);
  155|  1.04k|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (155:17): [True: 402, False: 639]
  ------------------
  156|    402|                memcpy(out->data[2], in->data[2], sz);
  157|  1.04k|        }
  158|  1.04k|    }
  159|  2.51k|}
dav1d_apply_grain_row_16bpc:
  167|  7.24k|{
  168|       |    // Synthesize grain for the affected planes
  169|  7.24k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  170|  7.24k|    const int ss_y = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  171|  7.24k|    const int ss_x = in->p.layout != DAV1D_PIXEL_LAYOUT_I444;
  172|  7.24k|    const int cpw = (out->p.w + ss_x) >> ss_x;
  173|  7.24k|    const int is_id = out->seq_hdr->mtrx == DAV1D_MC_IDENTITY;
  174|  7.24k|    pixel *const luma_src =
  175|  7.24k|        ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   37|  7.24k|#define FG_BLOCK_SIZE 32
  ------------------
  176|  7.24k|#if BITDEPTH != 8
  177|  7.24k|    const int bitdepth_max = (1 << out->p.bpc) - 1;
  178|  7.24k|#endif
  179|       |
  180|  7.24k|    if (data->num_y_points) {
  ------------------
  |  Branch (180:9): [True: 3.96k, False: 3.28k]
  ------------------
  181|  3.96k|        const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  3.96k|#define FG_BLOCK_SIZE 32
  ------------------
                      const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  3.96k|#define FG_BLOCK_SIZE 32
  ------------------
  182|  3.96k|        dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   37|  3.96k|#define FG_BLOCK_SIZE 32
  ------------------
  183|  3.96k|                         luma_src, out->stride[0], data,
  184|  3.96k|                         out->p.w, scaling[0], grain_lut[0], bh, row HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  3.96k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  185|  3.96k|    }
  186|       |
  187|  7.24k|    if (!data->num_uv_points[0] && !data->num_uv_points[1] &&
  ------------------
  |  Branch (187:9): [True: 5.00k, False: 2.24k]
  |  Branch (187:36): [True: 4.12k, False: 873]
  ------------------
  188|  4.12k|        !data->chroma_scaling_from_luma)
  ------------------
  |  Branch (188:9): [True: 3.17k, False: 951]
  ------------------
  189|  3.17k|    {
  190|  3.17k|        return;
  191|  3.17k|    }
  192|       |
  193|  4.06k|    const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  4.06k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  4.06k|#define FG_BLOCK_SIZE 32
  ------------------
  194|       |
  195|       |    // extend padding pixels
  196|  4.06k|    if (out->p.w & ss_x) {
  ------------------
  |  Branch (196:9): [True: 1.31k, False: 2.75k]
  ------------------
  197|  1.31k|        pixel *ptr = luma_src;
  198|  29.0k|        for (int y = 0; y < bh; y++) {
  ------------------
  |  Branch (198:25): [True: 27.7k, False: 1.31k]
  ------------------
  199|  27.7k|            ptr[out->p.w] = ptr[out->p.w - 1];
  200|  27.7k|            ptr += PXSTRIDE(in->stride[0]) << ss_y;
  201|  27.7k|        }
  202|  1.31k|    }
  203|       |
  204|  4.06k|    const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   37|  4.06k|#define FG_BLOCK_SIZE 32
  ------------------
  205|  4.06k|    if (data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (205:9): [True: 951, False: 3.11k]
  ------------------
  206|  2.85k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (206:26): [True: 1.90k, False: 951]
  ------------------
  207|  1.90k|            dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  208|  1.90k|                                                ((const pixel *) in->data[1 + pl]) + uv_off,
  209|  1.90k|                                                in->stride[1], data, cpw,
  210|  1.90k|                                                scaling[0], grain_lut[1 + pl],
  211|  1.90k|                                                bh, row, luma_src, in->stride[0],
  212|  1.90k|                                                pl, is_id HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  1.90k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  213|  3.11k|    } else {
  214|  9.35k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (214:26): [True: 6.23k, False: 3.11k]
  ------------------
  215|  6.23k|            if (data->num_uv_points[pl])
  ------------------
  |  Branch (215:17): [True: 3.55k, False: 2.68k]
  ------------------
  216|  3.55k|                dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  217|  3.55k|                                                    ((const pixel *) in->data[1 + pl]) + uv_off,
  218|  3.55k|                                                    in->stride[1], data, cpw,
  219|  3.55k|                                                    scaling[1 + pl], grain_lut[1 + pl],
  220|  3.55k|                                                    bh, row, luma_src, in->stride[0],
  221|  3.55k|                                                    pl, is_id HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  3.55k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  222|  3.11k|    }
  223|  4.06k|}
dav1d_apply_grain_16bpc:
  228|  2.51k|{
  229|  2.51k|    ALIGN_STK_16(entry, grain_lut, 3,[GRAIN_HEIGHT + 1][GRAIN_WIDTH]);
  ------------------
  |  |  100|  2.51k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  2.51k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  230|       |#if ARCH_X86_64 && BITDEPTH == 8
  231|       |    ALIGN_STK_64(uint8_t, scaling, 3,[SCALING_SIZE]);
  232|       |#else
  233|  2.51k|    uint8_t scaling[3][SCALING_SIZE];
  234|  2.51k|#endif
  235|  2.51k|    const int rows = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  2.51k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int rows = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  2.51k|#define FG_BLOCK_SIZE 32
  ------------------
  236|       |
  237|  2.51k|    bitfn(dav1d_prep_grain)(dsp, out, in, scaling, grain_lut);
  ------------------
  |  |   77|  2.51k|#define bitfn(x) x##_16bpc
  ------------------
  238|  9.75k|    for (int row = 0; row < rows; row++)
  ------------------
  |  Branch (238:23): [True: 7.24k, False: 2.51k]
  ------------------
  239|  7.24k|        bitfn(dav1d_apply_grain_row)(dsp, out, in, scaling, grain_lut, row);
  ------------------
  |  |   77|  7.24k|#define bitfn(x) x##_16bpc
  ------------------
  240|  2.51k|}

dav1d_film_grain_dsp_init_8bpc:
  425|  3.66k|COLD void bitfn(dav1d_film_grain_dsp_init)(Dav1dFilmGrainDSPContext *const c) {
  426|  3.66k|    c->generate_grain_y = generate_grain_y_c;
  427|  3.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = generate_grain_uv_420_c;
  428|  3.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = generate_grain_uv_422_c;
  429|  3.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = generate_grain_uv_444_c;
  430|       |
  431|  3.66k|    c->fgy_32x32xn = fgy_32x32xn_c;
  432|  3.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = fguv_32x32xn_420_c;
  433|  3.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = fguv_32x32xn_422_c;
  434|  3.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = fguv_32x32xn_444_c;
  435|       |
  436|  3.66k|#if HAVE_ASM
  437|       |#if ARCH_AARCH64 || ARCH_ARM
  438|       |    film_grain_dsp_init_arm(c);
  439|       |#elif ARCH_X86
  440|       |    film_grain_dsp_init_x86(c);
  441|       |#elif ARCH_RISCV
  442|       |    film_grain_dsp_init_riscv(c);
  443|       |#endif
  444|  3.66k|#endif
  445|  3.66k|}
dav1d_film_grain_dsp_init_16bpc:
  425|  4.99k|COLD void bitfn(dav1d_film_grain_dsp_init)(Dav1dFilmGrainDSPContext *const c) {
  426|  4.99k|    c->generate_grain_y = generate_grain_y_c;
  427|  4.99k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = generate_grain_uv_420_c;
  428|  4.99k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = generate_grain_uv_422_c;
  429|  4.99k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = generate_grain_uv_444_c;
  430|       |
  431|  4.99k|    c->fgy_32x32xn = fgy_32x32xn_c;
  432|  4.99k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = fguv_32x32xn_420_c;
  433|  4.99k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = fguv_32x32xn_422_c;
  434|  4.99k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = fguv_32x32xn_444_c;
  435|       |
  436|  4.99k|#if HAVE_ASM
  437|       |#if ARCH_AARCH64 || ARCH_ARM
  438|       |    film_grain_dsp_init_arm(c);
  439|       |#elif ARCH_X86
  440|       |    film_grain_dsp_init_x86(c);
  441|       |#elif ARCH_RISCV
  442|       |    film_grain_dsp_init_riscv(c);
  443|       |#endif
  444|  4.99k|#endif
  445|  4.99k|}

dav1d_init_get_bits:
   38|   127k|{
   39|   127k|    assert(sz);
  ------------------
  |  Branch (39:5): [True: 127k, False: 0]
  ------------------
   40|   127k|    c->ptr = c->ptr_start = data;
   41|   127k|    c->ptr_end = &c->ptr_start[sz];
   42|   127k|    c->state = 0;
   43|   127k|    c->bits_left = 0;
   44|   127k|    c->error = 0;
   45|   127k|}
dav1d_get_bit:
   47|  3.18M|unsigned dav1d_get_bit(GetBits *const c) {
   48|  3.18M|    if (!c->bits_left) {
  ------------------
  |  Branch (48:9): [True: 469k, False: 2.71M]
  ------------------
   49|   469k|        if (c->ptr >= c->ptr_end) {
  ------------------
  |  Branch (49:13): [True: 5.97k, False: 463k]
  ------------------
   50|  5.97k|            c->error = 1;
   51|   463k|        } else {
   52|   463k|            const unsigned state = *c->ptr++;
   53|   463k|            c->bits_left = 7;
   54|   463k|            c->state = (uint64_t) state << 57;
   55|   463k|            return state >> 7;
   56|   463k|        }
   57|   469k|    }
   58|       |
   59|  2.71M|    const uint64_t state = c->state;
   60|  2.71M|    c->bits_left--;
   61|  2.71M|    c->state = state << 1;
   62|  2.71M|    return (unsigned) (state >> 63);
   63|  3.18M|}
dav1d_get_uleb128:
   95|  60.9k|unsigned dav1d_get_uleb128(GetBits *const c) {
   96|  60.9k|    uint64_t val = 0;
   97|  60.9k|    unsigned i = 0, more;
   98|       |
   99|  65.5k|    do {
  100|  65.5k|        const int v = dav1d_get_bits(c, 8);
  101|  65.5k|        more = v & 0x80;
  102|  65.5k|        val |= ((uint64_t) (v & 0x7F)) << i;
  103|  65.5k|        i += 7;
  104|  65.5k|    } while (more && i < 56);
  ------------------
  |  Branch (104:14): [True: 4.93k, False: 60.5k]
  |  Branch (104:22): [True: 4.57k, False: 355]
  ------------------
  105|       |
  106|  60.9k|    if (val > UINT32_MAX || more) {
  ------------------
  |  Branch (106:9): [True: 265, False: 60.6k]
  |  Branch (106:29): [True: 326, False: 60.3k]
  ------------------
  107|    591|        c->error = 1;
  108|    591|        return 0;
  109|    591|    }
  110|       |
  111|  60.3k|    return (unsigned) val;
  112|  60.9k|}
dav1d_get_uniform:
  114|   133k|unsigned dav1d_get_uniform(GetBits *const c, const unsigned max) {
  115|       |    // Output in range [0..max-1]
  116|       |    // max must be > 1, or else nothing is read from the bitstream
  117|   133k|    assert(max > 1);
  ------------------
  |  Branch (117:5): [True: 133k, False: 0]
  ------------------
  118|   133k|    const int l = ulog2(max) + 1;
  119|   133k|    assert(l > 1);
  ------------------
  |  Branch (119:5): [True: 133k, False: 0]
  ------------------
  120|   133k|    const unsigned m = (1U << l) - max;
  121|   133k|    const unsigned v = dav1d_get_bits(c, l - 1);
  122|   133k|    return v < m ? v : (v << 1) - m + dav1d_get_bit(c);
  ------------------
  |  Branch (122:12): [True: 128k, False: 5.37k]
  ------------------
  123|   133k|}
dav1d_get_vlc:
  125|    824|unsigned dav1d_get_vlc(GetBits *const c) {
  126|    824|    if (dav1d_get_bit(c))
  ------------------
  |  Branch (126:9): [True: 396, False: 428]
  ------------------
  127|    396|        return 0;
  128|       |
  129|    428|    int n_bits = 0;
  130|  9.60k|    do {
  131|  9.60k|        if (++n_bits == 32)
  ------------------
  |  Branch (131:13): [True: 199, False: 9.40k]
  ------------------
  132|    199|            return UINT32_MAX;
  133|  9.60k|    } while (!dav1d_get_bit(c));
  ------------------
  |  Branch (133:14): [True: 9.17k, False: 229]
  ------------------
  134|       |
  135|    229|    return ((1U << n_bits) - 1) + dav1d_get_bits(c, n_bits);
  136|    428|}
dav1d_get_bits_subexp:
  162|  39.0k|int dav1d_get_bits_subexp(GetBits *const c, const int ref, const unsigned n) {
  163|  39.0k|    return (int) get_bits_subexp_u(c, ref + (1 << n), 2 << n) - (1 << n);
  164|  39.0k|}
getbits.c:refill:
   65|   643k|static inline void refill(GetBits *const c, const int n) {
   66|   643k|    assert(c->bits_left >= 0 && c->bits_left < 32);
  ------------------
  |  Branch (66:5): [True: 643k, False: 0]
  |  Branch (66:5): [True: 643k, False: 0]
  ------------------
   67|   643k|    unsigned state = 0;
   68|   718k|    do {
   69|   718k|        if (c->ptr >= c->ptr_end) {
  ------------------
  |  Branch (69:13): [True: 18.9k, False: 699k]
  ------------------
   70|  18.9k|            c->error = 1;
   71|  18.9k|            if (state) break;
  ------------------
  |  Branch (71:17): [True: 1.61k, False: 17.3k]
  ------------------
   72|  17.3k|            return;
   73|  18.9k|        }
   74|   699k|        state = (state << 8) | *c->ptr++;
   75|   699k|        c->bits_left += 8;
   76|   699k|    } while (n > c->bits_left);
  ------------------
  |  Branch (76:14): [True: 75.2k, False: 624k]
  ------------------
   77|   625k|    c->state |= (uint64_t) state << (64 - c->bits_left);
   78|   625k|}
getbits.c:get_bits_subexp_u:
  140|  39.0k|{
  141|  39.0k|    unsigned v = 0;
  142|       |
  143|   118k|    for (int i = 0;; i++) {
  144|   118k|        const int b = i ? 3 + i - 1 : 3;
  ------------------
  |  Branch (144:23): [True: 79.9k, False: 39.0k]
  ------------------
  145|       |
  146|   118k|        if (n < v + 3 * (1 << b)) {
  ------------------
  |  Branch (146:13): [True: 5.57k, False: 113k]
  ------------------
  147|  5.57k|            v += dav1d_get_uniform(c, n - v + 1);
  148|  5.57k|            break;
  149|  5.57k|        }
  150|       |
  151|   113k|        if (!dav1d_get_bit(c)) {
  ------------------
  |  Branch (151:13): [True: 33.5k, False: 79.9k]
  ------------------
  152|  33.5k|            v += dav1d_get_bits(c, b);
  153|  33.5k|            break;
  154|  33.5k|        }
  155|       |
  156|  79.9k|        v += 1 << b;
  157|  79.9k|    }
  158|       |
  159|  39.0k|    return ref * 2 <= n ? inv_recenter(ref, v) : n - inv_recenter(n - ref, v);
  ------------------
  |  Branch (159:12): [True: 34.3k, False: 4.75k]
  ------------------
  160|  39.0k|}

obu.c:dav1d_bytealign_get_bits:
   52|   125k|static inline void dav1d_bytealign_get_bits(GetBits *c) {
   53|       |    // bits_left is never more than 7, because it is only incremented
   54|       |    // by refill(), called by dav1d_get_bits and that never reads more
   55|       |    // than 7 bits more than it needs.
   56|       |    //
   57|       |    // If this wasn't true, we would need to work out how many bits to
   58|       |    // discard (bits_left % 8), subtract that from bits_left and then
   59|       |    // shift state right by that amount.
   60|   125k|    assert(c->bits_left <= 7);
  ------------------
  |  Branch (60:5): [True: 125k, False: 0]
  ------------------
   61|       |
   62|   125k|    c->bits_left = 0;
   63|   125k|    c->state = 0;
   64|   125k|}

dav1d_init_intra_edge_tree:
  126|      1|COLD void dav1d_init_intra_edge_tree(void) {
  127|       |    // This function is guaranteed to be called only once
  128|      1|    struct ModeSelMem mem;
  129|       |
  130|      1|    mem.nwc[BL_128X128] = &nodes.branch_sb128[1];
  131|      1|    mem.nwc[BL_64X64] = &nodes.branch_sb128[1 + 4];
  132|      1|    mem.nwc[BL_32X32] = &nodes.branch_sb128[1 + 4 + 16];
  133|      1|    mem.nt = nodes.tip_sb128;
  134|      1|    init_mode_node(nodes.branch_sb128, BL_128X128, &mem, 1, 0);
  135|      1|    assert(mem.nwc[BL_128X128] == &nodes.branch_sb128[1 + 4]);
  ------------------
  |  Branch (135:5): [True: 1, False: 0]
  ------------------
  136|      1|    assert(mem.nwc[BL_64X64] == &nodes.branch_sb128[1 + 4 + 16]);
  ------------------
  |  Branch (136:5): [True: 1, False: 0]
  ------------------
  137|      1|    assert(mem.nwc[BL_32X32] == &nodes.branch_sb128[1 + 4 + 16 + 64]);
  ------------------
  |  Branch (137:5): [True: 1, False: 0]
  ------------------
  138|      1|    assert(mem.nt == &nodes.tip_sb128[256]);
  ------------------
  |  Branch (138:5): [True: 1, False: 0]
  ------------------
  139|       |
  140|      1|    mem.nwc[BL_128X128] = NULL;
  141|      1|    mem.nwc[BL_64X64] = &nodes.branch_sb64[1];
  142|      1|    mem.nwc[BL_32X32] = &nodes.branch_sb64[1 + 4];
  143|      1|    mem.nt = nodes.tip_sb64;
  144|      1|    init_mode_node(nodes.branch_sb64, BL_64X64, &mem, 1, 0);
  145|      1|    assert(mem.nwc[BL_64X64] == &nodes.branch_sb64[1 + 4]);
  ------------------
  |  Branch (145:5): [True: 1, False: 0]
  ------------------
  146|      1|    assert(mem.nwc[BL_32X32] == &nodes.branch_sb64[1 + 4 + 16]);
  ------------------
  |  Branch (146:5): [True: 1, False: 0]
  ------------------
  147|      1|    assert(mem.nt == &nodes.tip_sb64[64]);
  ------------------
  |  Branch (147:5): [True: 1, False: 0]
  ------------------
  148|      1|}
intra_edge.c:init_mode_node:
  101|    106|{
  102|    106|    init_edges(&nwc->node, bl,
  103|    106|               (top_has_right ? EDGE_ALL_TOP_HAS_RIGHT : 0) |
  ------------------
  |  Branch (103:17): [True: 73, False: 33]
  ------------------
  104|    106|               (left_has_bottom ? EDGE_ALL_LEFT_HAS_BOTTOM : 0));
  ------------------
  |  Branch (104:17): [True: 33, False: 73]
  ------------------
  105|    106|    if (bl == BL_16X16) {
  ------------------
  |  Branch (105:9): [True: 80, False: 26]
  ------------------
  106|    400|        for (int n = 0; n < 4; n++) {
  ------------------
  |  Branch (106:25): [True: 320, False: 80]
  ------------------
  107|    320|            EdgeTip *const nt = mem->nt++;
  108|    320|            nwc->split_offset[n] = PTR_OFFSET(nwc, nt);
  ------------------
  |  |   94|    320|#define PTR_OFFSET(a, b) ((uint16_t)((uintptr_t)(b) - (uintptr_t)(a)))
  ------------------
  109|    320|            init_edges(&nt->node, bl + 1,
  110|    320|                       ((n == 3 || (n == 1 && !top_has_right)) ? 0 :
  ------------------
  |  Branch (110:26): [True: 80, False: 240]
  |  Branch (110:37): [True: 80, False: 160]
  |  Branch (110:47): [True: 26, False: 54]
  ------------------
  111|    320|                        EDGE_ALL_TOP_HAS_RIGHT) |
  112|    320|                       (!(n == 0 || (n == 2 && left_has_bottom)) ? 0 :
  ------------------
  |  Branch (112:27): [True: 80, False: 240]
  |  Branch (112:38): [True: 80, False: 160]
  |  Branch (112:48): [True: 26, False: 54]
  ------------------
  113|    320|                        EDGE_ALL_LEFT_HAS_BOTTOM));
  114|    320|        }
  115|     80|    } else {
  116|    130|        for (int n = 0; n < 4; n++) {
  ------------------
  |  Branch (116:25): [True: 104, False: 26]
  ------------------
  117|    104|            EdgeBranch *const nwc_child = mem->nwc[bl]++;
  118|    104|            nwc->split_offset[n] = PTR_OFFSET(nwc, nwc_child);
  ------------------
  |  |   94|    104|#define PTR_OFFSET(a, b) ((uint16_t)((uintptr_t)(b) - (uintptr_t)(a)))
  ------------------
  119|    104|            init_mode_node(nwc_child, bl + 1, mem,
  120|    104|                           !(n == 3 || (n == 1 && !top_has_right)),
  ------------------
  |  Branch (120:30): [True: 26, False: 78]
  |  Branch (120:41): [True: 26, False: 52]
  |  Branch (120:51): [True: 7, False: 19]
  ------------------
  121|    104|                           n == 0 || (n == 2 && left_has_bottom));
  ------------------
  |  Branch (121:28): [True: 26, False: 78]
  |  Branch (121:39): [True: 26, False: 52]
  |  Branch (121:49): [True: 7, False: 19]
  ------------------
  122|    104|        }
  123|     26|    }
  124|    106|}
intra_edge.c:init_edges:
   58|    426|{
   59|    426|    node->o = edge_flags;
   60|    426|    node->h[0] = edge_flags | EDGE_ALL_LEFT_HAS_BOTTOM;
   61|    426|    node->v[0] = edge_flags | EDGE_ALL_TOP_HAS_RIGHT;
   62|       |
   63|    426|    if (bl == BL_8X8) {
  ------------------
  |  Branch (63:9): [True: 320, False: 106]
  ------------------
   64|    320|        EdgeTip *const nt = (EdgeTip *) node;
   65|       |
   66|    320|        node->h[1] = edge_flags & (EDGE_ALL_LEFT_HAS_BOTTOM |
   67|    320|                                   EDGE_I420_TOP_HAS_RIGHT);
   68|    320|        node->v[1] = edge_flags & (EDGE_ALL_TOP_HAS_RIGHT |
   69|    320|                                   EDGE_I420_LEFT_HAS_BOTTOM |
   70|    320|                                   EDGE_I422_LEFT_HAS_BOTTOM);
   71|       |
   72|    320|        nt->split[0] = (edge_flags & EDGE_ALL_TOP_HAS_RIGHT) |
   73|    320|                       EDGE_I422_LEFT_HAS_BOTTOM;
   74|    320|        nt->split[1] = edge_flags | EDGE_I444_TOP_HAS_RIGHT;
   75|    320|        nt->split[2] = edge_flags & (EDGE_I420_TOP_HAS_RIGHT |
   76|    320|                                     EDGE_I420_LEFT_HAS_BOTTOM |
   77|    320|                                     EDGE_I422_LEFT_HAS_BOTTOM);
   78|    320|    } else {
   79|    106|        EdgeBranch *const nwc = (EdgeBranch *) node;
   80|       |
   81|    106|        node->h[1] = edge_flags & EDGE_ALL_LEFT_HAS_BOTTOM;
   82|    106|        node->v[1] = edge_flags & EDGE_ALL_TOP_HAS_RIGHT;
   83|       |
   84|    106|        nwc->h4 = EDGE_ALL_LEFT_HAS_BOTTOM;
   85|    106|        nwc->v4 = EDGE_ALL_TOP_HAS_RIGHT;
   86|    106|        if (bl == BL_16X16) {
  ------------------
  |  Branch (86:13): [True: 80, False: 26]
  ------------------
   87|     80|            nwc->h4 |= edge_flags & EDGE_I420_TOP_HAS_RIGHT;
   88|     80|            nwc->v4 |= edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM |
   89|     80|                                     EDGE_I422_LEFT_HAS_BOTTOM);
   90|     80|        }
   91|    106|    }
   92|    426|}

recon_tmpl.c:sm_flag:
   95|  5.29M|static inline int sm_flag(const BlockContext *const b, const int idx) {
   96|  5.29M|    if (!b->intra[idx]) return 0;
  ------------------
  |  Branch (96:9): [True: 203k, False: 5.09M]
  ------------------
   97|  5.09M|    const enum IntraPredMode m = b->mode[idx];
   98|  5.09M|    return (m == SMOOTH_PRED || m == SMOOTH_H_PRED ||
  ------------------
  |  Branch (98:13): [True: 343k, False: 4.75M]
  |  Branch (98:33): [True: 113k, False: 4.63M]
  ------------------
   99|  4.63M|            m == SMOOTH_V_PRED) ? ANGLE_SMOOTH_EDGE_FLAG : 0;
  ------------------
  |  |   93|   555k|#define ANGLE_SMOOTH_EDGE_FLAG      512
  ------------------
  |  Branch (99:13): [True: 98.4k, False: 4.54M]
  ------------------
  100|  5.29M|}
recon_tmpl.c:sm_uv_flag:
  102|  3.66M|static inline int sm_uv_flag(const BlockContext *const b, const int idx) {
  103|  3.66M|    const enum IntraPredMode m = b->uvmode[idx];
  104|  3.66M|    return (m == SMOOTH_PRED || m == SMOOTH_H_PRED ||
  ------------------
  |  Branch (104:13): [True: 254k, False: 3.40M]
  |  Branch (104:33): [True: 95.8k, False: 3.31M]
  ------------------
  105|  3.31M|            m == SMOOTH_V_PRED) ? ANGLE_SMOOTH_EDGE_FLAG : 0;
  ------------------
  |  |   93|   421k|#define ANGLE_SMOOTH_EDGE_FLAG      512
  ------------------
  |  Branch (105:13): [True: 71.5k, False: 3.24M]
  ------------------
  106|  3.66M|}

dav1d_prepare_intra_edges_8bpc:
   86|  6.51M|{
   87|  6.51M|    const int bitdepth = bitdepth_from_max(bitdepth_max);
  ------------------
  |  |   58|  6.51M|#define bitdepth_from_max(x) 8
  ------------------
   88|  6.51M|    assert(y < h && x < w);
  ------------------
  |  Branch (88:5): [True: 6.51M, False: 0]
  |  Branch (88:5): [True: 6.51M, False: 0]
  ------------------
   89|       |
   90|  6.51M|    switch (mode) {
   91|   162k|    case VERT_PRED:
  ------------------
  |  Branch (91:5): [True: 162k, False: 6.34M]
  ------------------
   92|   468k|    case HOR_PRED:
  ------------------
  |  Branch (92:5): [True: 305k, False: 6.20M]
  ------------------
   93|   532k|    case DIAG_DOWN_LEFT_PRED:
  ------------------
  |  Branch (93:5): [True: 64.2k, False: 6.44M]
  ------------------
   94|   609k|    case DIAG_DOWN_RIGHT_PRED:
  ------------------
  |  Branch (94:5): [True: 77.5k, False: 6.43M]
  ------------------
   95|   685k|    case VERT_RIGHT_PRED:
  ------------------
  |  Branch (95:5): [True: 75.3k, False: 6.43M]
  ------------------
   96|   761k|    case HOR_DOWN_PRED:
  ------------------
  |  Branch (96:5): [True: 76.4k, False: 6.43M]
  ------------------
   97|   893k|    case HOR_UP_PRED:
  ------------------
  |  Branch (97:5): [True: 131k, False: 6.38M]
  ------------------
   98|   983k|    case VERT_LEFT_PRED: {
  ------------------
  |  Branch (98:5): [True: 90.3k, False: 6.42M]
  ------------------
   99|   983k|        *angle = av1_mode_to_angle_map[mode - VERT_PRED] + 3 * *angle;
  100|       |
  101|   983k|        if (*angle <= 90)
  ------------------
  |  Branch (101:13): [True: 280k, False: 703k]
  ------------------
  102|   280k|            mode = *angle < 90 && have_top ? Z1_PRED : VERT_PRED;
  ------------------
  |  Branch (102:20): [True: 182k, False: 98.2k]
  |  Branch (102:35): [True: 149k, False: 32.4k]
  ------------------
  103|   703k|        else if (*angle < 180)
  ------------------
  |  Branch (103:18): [True: 327k, False: 375k]
  ------------------
  104|   327k|            mode = Z2_PRED;
  105|   375k|        else
  106|   375k|            mode = *angle > 180 && have_left ? Z3_PRED : HOR_PRED;
  ------------------
  |  Branch (106:20): [True: 196k, False: 179k]
  |  Branch (106:36): [True: 193k, False: 3.28k]
  ------------------
  107|   983k|        break;
  108|   893k|    }
  109|  4.01M|    case DC_PRED:
  ------------------
  |  Branch (109:5): [True: 4.01M, False: 2.49M]
  ------------------
  110|  4.69M|    case PAETH_PRED:
  ------------------
  |  Branch (110:5): [True: 676k, False: 5.83M]
  ------------------
  111|  4.69M|        mode = av1_mode_conv[mode][have_left][have_top];
  112|  4.69M|        break;
  113|   833k|    default:
  ------------------
  |  Branch (113:5): [True: 833k, False: 5.67M]
  ------------------
  114|   833k|        break;
  115|  6.51M|    }
  116|       |
  117|  6.51M|    const pixel *dst_top;
  118|  6.51M|    if (have_top &&
  ------------------
  |  Branch (118:9): [True: 5.67M, False: 833k]
  ------------------
  119|  5.67M|        (av1_intra_prediction_edges[mode].needs_top ||
  ------------------
  |  Branch (119:10): [True: 5.36M, False: 315k]
  ------------------
  120|   315k|         av1_intra_prediction_edges[mode].needs_topleft ||
  ------------------
  |  Branch (120:10): [True: 158k, False: 156k]
  ------------------
  121|   156k|         (av1_intra_prediction_edges[mode].needs_left && !have_left)))
  ------------------
  |  Branch (121:11): [True: 156k, False: 0]
  |  Branch (121:58): [True: 5.96k, False: 150k]
  ------------------
  122|  5.52M|    {
  123|  5.52M|        if (prefilter_toplevel_sb_edge) {
  ------------------
  |  Branch (123:13): [True: 314k, False: 5.21M]
  ------------------
  124|   314k|            dst_top = &prefilter_toplevel_sb_edge[x * 4];
  125|  5.21M|        } else {
  126|  5.21M|            dst_top = &dst[-PXSTRIDE(stride)];
  ------------------
  |  |   53|  5.21M|#define PXSTRIDE(x) (x)
  ------------------
  127|  5.21M|        }
  128|  5.52M|    }
  129|       |
  130|  6.51M|    if (av1_intra_prediction_edges[mode].needs_left) {
  ------------------
  |  Branch (130:9): [True: 5.92M, False: 582k]
  ------------------
  131|  5.92M|        const int sz = th << 2;
  132|  5.92M|        pixel *const left = &topleft_out[-sz];
  133|       |
  134|  5.92M|        if (have_left) {
  ------------------
  |  Branch (134:13): [True: 5.90M, False: 21.7k]
  ------------------
  135|  5.90M|            const int px_have = imin(sz, (h - y) << 2);
  136|       |
  137|  63.2M|            for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (137:29): [True: 57.3M, False: 5.90M]
  ------------------
  138|  57.3M|                left[sz - 1 - i] = dst[PXSTRIDE(stride) * i - 1];
  ------------------
  |  |   53|  57.3M|#define PXSTRIDE(x) (x)
  ------------------
  139|  5.90M|            if (px_have < sz)
  ------------------
  |  Branch (139:17): [True: 124k, False: 5.78M]
  ------------------
  140|   124k|                pixel_set(left, left[sz - px_have], sz - px_have);
  ------------------
  |  |   48|   124k|#define pixel_set memset
  ------------------
  141|  5.90M|        } else {
  142|  21.7k|            pixel_set(left, have_top ? *dst_top : ((1 << bitdepth) >> 1) + 1, sz);
  ------------------
  |  |   48|  21.7k|#define pixel_set memset
  ------------------
  |  Branch (142:29): [True: 15.9k, False: 5.84k]
  ------------------
  143|  21.7k|        }
  144|       |
  145|  5.92M|        if (av1_intra_prediction_edges[mode].needs_bottomleft) {
  ------------------
  |  Branch (145:13): [True: 193k, False: 5.73M]
  ------------------
  146|   193k|            const int have_bottomleft = (!have_left || y + th >= h) ? 0 :
  ------------------
  |  Branch (146:42): [True: 0, False: 193k]
  |  Branch (146:56): [True: 40.1k, False: 153k]
  ------------------
  147|   193k|                                        (edge_flags & EDGE_I444_LEFT_HAS_BOTTOM);
  148|       |
  149|   193k|            if (have_bottomleft) {
  ------------------
  |  Branch (149:17): [True: 58.3k, False: 134k]
  ------------------
  150|  58.3k|                const int px_have = imin(sz, (h - y - th) << 2);
  151|       |
  152|   597k|                for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (152:33): [True: 539k, False: 58.3k]
  ------------------
  153|   539k|                    left[-(i + 1)] = dst[(sz + i) * PXSTRIDE(stride) - 1];
  ------------------
  |  |   53|   539k|#define PXSTRIDE(x) (x)
  ------------------
  154|  58.3k|                if (px_have < sz)
  ------------------
  |  Branch (154:21): [True: 2.50k, False: 55.8k]
  ------------------
  155|  2.50k|                    pixel_set(left - sz, left[-px_have], sz - px_have);
  ------------------
  |  |   48|  2.50k|#define pixel_set memset
  ------------------
  156|   134k|            } else {
  157|   134k|                pixel_set(left - sz, left[0], sz);
  ------------------
  |  |   48|   134k|#define pixel_set memset
  ------------------
  158|   134k|            }
  159|   193k|        }
  160|  5.92M|    }
  161|       |
  162|  6.51M|    if (av1_intra_prediction_edges[mode].needs_top) {
  ------------------
  |  Branch (162:9): [True: 5.60M, False: 902k]
  ------------------
  163|  5.60M|        const int sz = tw << 2;
  164|  5.60M|        pixel *const top = &topleft_out[1];
  165|       |
  166|  5.60M|        if (have_top) {
  ------------------
  |  Branch (166:13): [True: 5.36M, False: 245k]
  ------------------
  167|  5.36M|            const int px_have = imin(sz, (w - x) << 2);
  168|  5.36M|            pixel_copy(top, dst_top, px_have);
  ------------------
  |  |   47|  5.36M|#define pixel_copy memcpy
  ------------------
  169|  5.36M|            if (px_have < sz)
  ------------------
  |  Branch (169:17): [True: 102k, False: 5.26M]
  ------------------
  170|   102k|                pixel_set(top + px_have, top[px_have - 1], sz - px_have);
  ------------------
  |  |   48|   102k|#define pixel_set memset
  ------------------
  171|  5.36M|        } else {
  172|   245k|            pixel_set(top, have_left ? dst[-1] : ((1 << bitdepth) >> 1) - 1, sz);
  ------------------
  |  |   48|   245k|#define pixel_set memset
  ------------------
  |  Branch (172:28): [True: 239k, False: 6.03k]
  ------------------
  173|   245k|        }
  174|       |
  175|  5.60M|        if (av1_intra_prediction_edges[mode].needs_topright) {
  ------------------
  |  Branch (175:13): [True: 149k, False: 5.45M]
  ------------------
  176|   149k|            const int have_topright = (!have_top || x + tw >= w) ? 0 :
  ------------------
  |  Branch (176:40): [True: 0, False: 149k]
  |  Branch (176:53): [True: 2.79k, False: 146k]
  ------------------
  177|   149k|                                      (edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  178|       |
  179|   149k|            if (have_topright) {
  ------------------
  |  Branch (179:17): [True: 91.9k, False: 57.7k]
  ------------------
  180|  91.9k|                const int px_have = imin(sz, (w - x - tw) << 2);
  181|       |
  182|  91.9k|                pixel_copy(top + sz, &dst_top[sz], px_have);
  ------------------
  |  |   47|  91.9k|#define pixel_copy memcpy
  ------------------
  183|  91.9k|                if (px_have < sz)
  ------------------
  |  Branch (183:21): [True: 677, False: 91.2k]
  ------------------
  184|    677|                    pixel_set(top + sz + px_have, top[sz + px_have - 1],
  ------------------
  |  |   48|    677|#define pixel_set memset
  ------------------
  185|    677|                              sz - px_have);
  186|  91.9k|            } else {
  187|  57.7k|                pixel_set(top + sz, top[sz - 1], sz);
  ------------------
  |  |   48|  57.7k|#define pixel_set memset
  ------------------
  188|  57.7k|            }
  189|   149k|        }
  190|  5.60M|    }
  191|       |
  192|  6.51M|    if (av1_intra_prediction_edges[mode].needs_topleft) {
  ------------------
  |  Branch (192:9): [True: 1.50M, False: 5.00M]
  ------------------
  193|  1.50M|        if (have_left)
  ------------------
  |  Branch (193:13): [True: 1.49M, False: 11.1k]
  ------------------
  194|  1.49M|            *topleft_out = have_top ? dst_top[-1] : dst[-1];
  ------------------
  |  Branch (194:28): [True: 1.35M, False: 137k]
  ------------------
  195|  11.1k|        else
  196|  11.1k|            *topleft_out = have_top ? *dst_top : (1 << bitdepth) >> 1;
  ------------------
  |  Branch (196:28): [True: 8.52k, False: 2.63k]
  ------------------
  197|       |
  198|  1.50M|        if (mode == Z2_PRED && tw + th >= 6 && filter_edge)
  ------------------
  |  Branch (198:13): [True: 327k, False: 1.17M]
  |  Branch (198:32): [True: 135k, False: 192k]
  |  Branch (198:48): [True: 26.4k, False: 108k]
  ------------------
  199|  26.4k|            *topleft_out = ((topleft_out[-1] + topleft_out[1]) * 5 +
  200|  26.4k|                            topleft_out[0] * 6 + 8) >> 4;
  201|  1.50M|    }
  202|       |
  203|  6.51M|    return mode;
  204|  6.51M|}
dav1d_prepare_intra_edges_16bpc:
   86|  7.10M|{
   87|  7.10M|    const int bitdepth = bitdepth_from_max(bitdepth_max);
  ------------------
  |  |   75|  7.10M|#define bitdepth_from_max(bitdepth_max) (32 - clz(bitdepth_max))
  ------------------
   88|  7.10M|    assert(y < h && x < w);
  ------------------
  |  Branch (88:5): [True: 7.10M, False: 0]
  |  Branch (88:5): [True: 7.10M, False: 0]
  ------------------
   89|       |
   90|  7.10M|    switch (mode) {
   91|   219k|    case VERT_PRED:
  ------------------
  |  Branch (91:5): [True: 219k, False: 6.88M]
  ------------------
   92|   608k|    case HOR_PRED:
  ------------------
  |  Branch (92:5): [True: 388k, False: 6.71M]
  ------------------
   93|   706k|    case DIAG_DOWN_LEFT_PRED:
  ------------------
  |  Branch (93:5): [True: 97.7k, False: 7.00M]
  ------------------
   94|   787k|    case DIAG_DOWN_RIGHT_PRED:
  ------------------
  |  Branch (94:5): [True: 80.8k, False: 7.01M]
  ------------------
   95|   851k|    case VERT_RIGHT_PRED:
  ------------------
  |  Branch (95:5): [True: 64.1k, False: 7.03M]
  ------------------
   96|  1.00M|    case HOR_DOWN_PRED:
  ------------------
  |  Branch (96:5): [True: 152k, False: 6.94M]
  ------------------
   97|  1.21M|    case HOR_UP_PRED:
  ------------------
  |  Branch (97:5): [True: 210k, False: 6.88M]
  ------------------
   98|  1.34M|    case VERT_LEFT_PRED: {
  ------------------
  |  Branch (98:5): [True: 126k, False: 6.97M]
  ------------------
   99|  1.34M|        *angle = av1_mode_to_angle_map[mode - VERT_PRED] + 3 * *angle;
  100|       |
  101|  1.34M|        if (*angle <= 90)
  ------------------
  |  Branch (101:13): [True: 392k, False: 948k]
  ------------------
  102|   392k|            mode = *angle < 90 && have_top ? Z1_PRED : VERT_PRED;
  ------------------
  |  Branch (102:20): [True: 258k, False: 133k]
  |  Branch (102:35): [True: 198k, False: 60.2k]
  ------------------
  103|   948k|        else if (*angle < 180)
  ------------------
  |  Branch (103:18): [True: 429k, False: 518k]
  ------------------
  104|   429k|            mode = Z2_PRED;
  105|   518k|        else
  106|   518k|            mode = *angle > 180 && have_left ? Z3_PRED : HOR_PRED;
  ------------------
  |  Branch (106:20): [True: 293k, False: 225k]
  |  Branch (106:36): [True: 282k, False: 10.5k]
  ------------------
  107|  1.34M|        break;
  108|  1.21M|    }
  109|  4.17M|    case DC_PRED:
  ------------------
  |  Branch (109:5): [True: 4.17M, False: 2.92M]
  ------------------
  110|  4.84M|    case PAETH_PRED:
  ------------------
  |  Branch (110:5): [True: 670k, False: 6.42M]
  ------------------
  111|  4.84M|        mode = av1_mode_conv[mode][have_left][have_top];
  112|  4.84M|        break;
  113|   914k|    default:
  ------------------
  |  Branch (113:5): [True: 914k, False: 6.18M]
  ------------------
  114|   914k|        break;
  115|  7.10M|    }
  116|       |
  117|  7.10M|    const pixel *dst_top;
  118|  7.10M|    if (have_top &&
  ------------------
  |  Branch (118:9): [True: 5.66M, False: 1.43M]
  ------------------
  119|  5.66M|        (av1_intra_prediction_edges[mode].needs_top ||
  ------------------
  |  Branch (119:10): [True: 5.27M, False: 390k]
  ------------------
  120|   390k|         av1_intra_prediction_edges[mode].needs_topleft ||
  ------------------
  |  Branch (120:10): [True: 202k, False: 188k]
  ------------------
  121|   188k|         (av1_intra_prediction_edges[mode].needs_left && !have_left)))
  ------------------
  |  Branch (121:11): [True: 188k, False: 0]
  |  Branch (121:58): [True: 12.0k, False: 176k]
  ------------------
  122|  5.49M|    {
  123|  5.49M|        if (prefilter_toplevel_sb_edge) {
  ------------------
  |  Branch (123:13): [True: 345k, False: 5.14M]
  ------------------
  124|   345k|            dst_top = &prefilter_toplevel_sb_edge[x * 4];
  125|  5.14M|        } else {
  126|  5.14M|            dst_top = &dst[-PXSTRIDE(stride)];
  127|  5.14M|        }
  128|  5.49M|    }
  129|       |
  130|  7.10M|    if (av1_intra_prediction_edges[mode].needs_left) {
  ------------------
  |  Branch (130:9): [True: 6.32M, False: 771k]
  ------------------
  131|  6.32M|        const int sz = th << 2;
  132|  6.32M|        pixel *const left = &topleft_out[-sz];
  133|       |
  134|  6.32M|        if (have_left) {
  ------------------
  |  Branch (134:13): [True: 6.28M, False: 46.2k]
  ------------------
  135|  6.28M|            const int px_have = imin(sz, (h - y) << 2);
  136|       |
  137|  83.5M|            for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (137:29): [True: 77.3M, False: 6.28M]
  ------------------
  138|  77.3M|                left[sz - 1 - i] = dst[PXSTRIDE(stride) * i - 1];
  139|  6.28M|            if (px_have < sz)
  ------------------
  |  Branch (139:17): [True: 269k, False: 6.01M]
  ------------------
  140|   269k|                pixel_set(left, left[sz - px_have], sz - px_have);
  141|  6.28M|        } else {
  142|  46.2k|            pixel_set(left, have_top ? *dst_top : ((1 << bitdepth) >> 1) + 1, sz);
  ------------------
  |  Branch (142:29): [True: 38.3k, False: 7.87k]
  ------------------
  143|  46.2k|        }
  144|       |
  145|  6.32M|        if (av1_intra_prediction_edges[mode].needs_bottomleft) {
  ------------------
  |  Branch (145:13): [True: 282k, False: 6.04M]
  ------------------
  146|   282k|            const int have_bottomleft = (!have_left || y + th >= h) ? 0 :
  ------------------
  |  Branch (146:42): [True: 0, False: 282k]
  |  Branch (146:56): [True: 78.6k, False: 203k]
  ------------------
  147|   282k|                                        (edge_flags & EDGE_I444_LEFT_HAS_BOTTOM);
  148|       |
  149|   282k|            if (have_bottomleft) {
  ------------------
  |  Branch (149:17): [True: 83.1k, False: 199k]
  ------------------
  150|  83.1k|                const int px_have = imin(sz, (h - y - th) << 2);
  151|       |
  152|   955k|                for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (152:33): [True: 872k, False: 83.1k]
  ------------------
  153|   872k|                    left[-(i + 1)] = dst[(sz + i) * PXSTRIDE(stride) - 1];
  154|  83.1k|                if (px_have < sz)
  ------------------
  |  Branch (154:21): [True: 5.99k, False: 77.1k]
  ------------------
  155|  5.99k|                    pixel_set(left - sz, left[-px_have], sz - px_have);
  156|   199k|            } else {
  157|   199k|                pixel_set(left - sz, left[0], sz);
  158|   199k|            }
  159|   282k|        }
  160|  6.32M|    }
  161|       |
  162|  7.10M|    if (av1_intra_prediction_edges[mode].needs_top) {
  ------------------
  |  Branch (162:9): [True: 5.74M, False: 1.35M]
  ------------------
  163|  5.74M|        const int sz = tw << 2;
  164|  5.74M|        pixel *const top = &topleft_out[1];
  165|       |
  166|  5.74M|        if (have_top) {
  ------------------
  |  Branch (166:13): [True: 5.27M, False: 464k]
  ------------------
  167|  5.27M|            const int px_have = imin(sz, (w - x) << 2);
  168|  5.27M|            pixel_copy(top, dst_top, px_have);
  ------------------
  |  |   65|  5.27M|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
  169|  5.27M|            if (px_have < sz)
  ------------------
  |  Branch (169:17): [True: 217k, False: 5.06M]
  ------------------
  170|   217k|                pixel_set(top + px_have, top[px_have - 1], sz - px_have);
  171|  5.27M|        } else {
  172|   464k|            pixel_set(top, have_left ? dst[-1] : ((1 << bitdepth) >> 1) - 1, sz);
  ------------------
  |  Branch (172:28): [True: 456k, False: 7.95k]
  ------------------
  173|   464k|        }
  174|       |
  175|  5.74M|        if (av1_intra_prediction_edges[mode].needs_topright) {
  ------------------
  |  Branch (175:13): [True: 198k, False: 5.54M]
  ------------------
  176|   198k|            const int have_topright = (!have_top || x + tw >= w) ? 0 :
  ------------------
  |  Branch (176:40): [True: 0, False: 198k]
  |  Branch (176:53): [True: 5.72k, False: 193k]
  ------------------
  177|   198k|                                      (edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  178|       |
  179|   198k|            if (have_topright) {
  ------------------
  |  Branch (179:17): [True: 121k, False: 77.4k]
  ------------------
  180|   121k|                const int px_have = imin(sz, (w - x - tw) << 2);
  181|       |
  182|   121k|                pixel_copy(top + sz, &dst_top[sz], px_have);
  ------------------
  |  |   65|   121k|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
  183|   121k|                if (px_have < sz)
  ------------------
  |  Branch (183:21): [True: 999, False: 120k]
  ------------------
  184|    999|                    pixel_set(top + sz + px_have, top[sz + px_have - 1],
  185|    999|                              sz - px_have);
  186|   121k|            } else {
  187|  77.4k|                pixel_set(top + sz, top[sz - 1], sz);
  188|  77.4k|            }
  189|   198k|        }
  190|  5.74M|    }
  191|       |
  192|  7.10M|    if (av1_intra_prediction_edges[mode].needs_topleft) {
  ------------------
  |  Branch (192:9): [True: 1.59M, False: 5.50M]
  ------------------
  193|  1.59M|        if (have_left)
  ------------------
  |  Branch (193:13): [True: 1.56M, False: 24.8k]
  ------------------
  194|  1.56M|            *topleft_out = have_top ? dst_top[-1] : dst[-1];
  ------------------
  |  Branch (194:28): [True: 1.32M, False: 239k]
  ------------------
  195|  24.8k|        else
  196|  24.8k|            *topleft_out = have_top ? *dst_top : (1 << bitdepth) >> 1;
  ------------------
  |  Branch (196:28): [True: 23.1k, False: 1.73k]
  ------------------
  197|       |
  198|  1.59M|        if (mode == Z2_PRED && tw + th >= 6 && filter_edge)
  ------------------
  |  Branch (198:13): [True: 429k, False: 1.16M]
  |  Branch (198:32): [True: 207k, False: 222k]
  |  Branch (198:48): [True: 58.3k, False: 149k]
  ------------------
  199|  58.3k|            *topleft_out = ((topleft_out[-1] + topleft_out[1]) * 5 +
  200|  58.3k|                            topleft_out[0] * 6 + 8) >> 4;
  201|  1.59M|    }
  202|       |
  203|  7.10M|    return mode;
  204|  7.10M|}

dav1d_intra_pred_dsp_init_8bpc:
  744|  3.66k|COLD void bitfn(dav1d_intra_pred_dsp_init)(Dav1dIntraPredDSPContext *const c) {
  745|  3.66k|    c->intra_pred[DC_PRED      ] = ipred_dc_c;
  746|  3.66k|    c->intra_pred[DC_128_PRED  ] = ipred_dc_128_c;
  747|  3.66k|    c->intra_pred[TOP_DC_PRED  ] = ipred_dc_top_c;
  748|  3.66k|    c->intra_pred[LEFT_DC_PRED ] = ipred_dc_left_c;
  749|  3.66k|    c->intra_pred[HOR_PRED     ] = ipred_h_c;
  750|  3.66k|    c->intra_pred[VERT_PRED    ] = ipred_v_c;
  751|  3.66k|    c->intra_pred[PAETH_PRED   ] = ipred_paeth_c;
  752|  3.66k|    c->intra_pred[SMOOTH_PRED  ] = ipred_smooth_c;
  753|  3.66k|    c->intra_pred[SMOOTH_V_PRED] = ipred_smooth_v_c;
  754|  3.66k|    c->intra_pred[SMOOTH_H_PRED] = ipred_smooth_h_c;
  755|  3.66k|    c->intra_pred[Z1_PRED      ] = ipred_z1_c;
  756|  3.66k|    c->intra_pred[Z2_PRED      ] = ipred_z2_c;
  757|  3.66k|    c->intra_pred[Z3_PRED      ] = ipred_z3_c;
  758|  3.66k|    c->intra_pred[FILTER_PRED  ] = ipred_filter_c;
  759|       |
  760|  3.66k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I420 - 1] = cfl_ac_420_c;
  761|  3.66k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I422 - 1] = cfl_ac_422_c;
  762|  3.66k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I444 - 1] = cfl_ac_444_c;
  763|       |
  764|  3.66k|    c->cfl_pred[DC_PRED     ] = ipred_cfl_c;
  765|  3.66k|    c->cfl_pred[DC_128_PRED ] = ipred_cfl_128_c;
  766|  3.66k|    c->cfl_pred[TOP_DC_PRED ] = ipred_cfl_top_c;
  767|  3.66k|    c->cfl_pred[LEFT_DC_PRED] = ipred_cfl_left_c;
  768|       |
  769|  3.66k|    c->pal_pred = pal_pred_c;
  770|       |
  771|  3.66k|#if HAVE_ASM
  772|       |#if ARCH_AARCH64 || ARCH_ARM
  773|       |    intra_pred_dsp_init_arm(c);
  774|       |#elif ARCH_RISCV
  775|       |    intra_pred_dsp_init_riscv(c);
  776|       |#elif ARCH_X86
  777|       |    intra_pred_dsp_init_x86(c);
  778|       |#elif ARCH_LOONGARCH64
  779|       |    intra_pred_dsp_init_loongarch(c);
  780|       |#endif
  781|  3.66k|#endif
  782|  3.66k|}
dav1d_intra_pred_dsp_init_16bpc:
  744|  4.99k|COLD void bitfn(dav1d_intra_pred_dsp_init)(Dav1dIntraPredDSPContext *const c) {
  745|  4.99k|    c->intra_pred[DC_PRED      ] = ipred_dc_c;
  746|  4.99k|    c->intra_pred[DC_128_PRED  ] = ipred_dc_128_c;
  747|  4.99k|    c->intra_pred[TOP_DC_PRED  ] = ipred_dc_top_c;
  748|  4.99k|    c->intra_pred[LEFT_DC_PRED ] = ipred_dc_left_c;
  749|  4.99k|    c->intra_pred[HOR_PRED     ] = ipred_h_c;
  750|  4.99k|    c->intra_pred[VERT_PRED    ] = ipred_v_c;
  751|  4.99k|    c->intra_pred[PAETH_PRED   ] = ipred_paeth_c;
  752|  4.99k|    c->intra_pred[SMOOTH_PRED  ] = ipred_smooth_c;
  753|  4.99k|    c->intra_pred[SMOOTH_V_PRED] = ipred_smooth_v_c;
  754|  4.99k|    c->intra_pred[SMOOTH_H_PRED] = ipred_smooth_h_c;
  755|  4.99k|    c->intra_pred[Z1_PRED      ] = ipred_z1_c;
  756|  4.99k|    c->intra_pred[Z2_PRED      ] = ipred_z2_c;
  757|  4.99k|    c->intra_pred[Z3_PRED      ] = ipred_z3_c;
  758|  4.99k|    c->intra_pred[FILTER_PRED  ] = ipred_filter_c;
  759|       |
  760|  4.99k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I420 - 1] = cfl_ac_420_c;
  761|  4.99k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I422 - 1] = cfl_ac_422_c;
  762|  4.99k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I444 - 1] = cfl_ac_444_c;
  763|       |
  764|  4.99k|    c->cfl_pred[DC_PRED     ] = ipred_cfl_c;
  765|  4.99k|    c->cfl_pred[DC_128_PRED ] = ipred_cfl_128_c;
  766|  4.99k|    c->cfl_pred[TOP_DC_PRED ] = ipred_cfl_top_c;
  767|  4.99k|    c->cfl_pred[LEFT_DC_PRED] = ipred_cfl_left_c;
  768|       |
  769|  4.99k|    c->pal_pred = pal_pred_c;
  770|       |
  771|  4.99k|#if HAVE_ASM
  772|       |#if ARCH_AARCH64 || ARCH_ARM
  773|       |    intra_pred_dsp_init_arm(c);
  774|       |#elif ARCH_RISCV
  775|       |    intra_pred_dsp_init_riscv(c);
  776|       |#elif ARCH_X86
  777|       |    intra_pred_dsp_init_x86(c);
  778|       |#elif ARCH_LOONGARCH64
  779|       |    intra_pred_dsp_init_loongarch(c);
  780|       |#endif
  781|  4.99k|#endif
  782|  4.99k|}

itx_1d.c:inv_dct4_1d_internal_c:
   68|  6.02M|{
   69|  6.02M|    assert(stride > 0);
  ------------------
  |  Branch (69:5): [True: 6.02M, False: 0]
  ------------------
   70|  6.02M|    const int in0 = c[0 * stride], in1 = c[1 * stride];
   71|       |
   72|  6.02M|    int t0, t1, t2, t3;
   73|  6.02M|    if (tx64) {
  ------------------
  |  Branch (73:9): [True: 1.93M, False: 4.08M]
  ------------------
   74|  1.93M|        t0 = t1 = (in0 * 181 + 128) >> 8;
   75|  1.93M|        t2 = (in1 * 1567 + 2048) >> 12;
   76|  1.93M|        t3 = (in1 * 3784 + 2048) >> 12;
   77|  4.08M|    } else {
   78|  4.08M|        const int in2 = c[2 * stride], in3 = c[3 * stride];
   79|       |
   80|  4.08M|        t0 = ((in0 + in2) * 181 + 128) >> 8;
   81|  4.08M|        t1 = ((in0 - in2) * 181 + 128) >> 8;
   82|  4.08M|        t2 = ((in1 *  1567         - in3 * (3784 - 4096) + 2048) >> 12) - in3;
   83|  4.08M|        t3 = ((in1 * (3784 - 4096) + in3 *  1567         + 2048) >> 12) + in1;
   84|  4.08M|    }
   85|       |
   86|  6.02M|    c[0 * stride] = CLIP(t0 + t3);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
   87|  6.02M|    c[1 * stride] = CLIP(t1 + t2);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
   88|  6.02M|    c[2 * stride] = CLIP(t1 - t2);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
   89|  6.02M|    c[3 * stride] = CLIP(t0 - t3);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
   90|  6.02M|}
itx_1d.c:inv_dct8_1d_internal_c:
  101|  6.02M|{
  102|  6.02M|    assert(stride > 0);
  ------------------
  |  Branch (102:5): [True: 6.02M, False: 0]
  ------------------
  103|  6.02M|    inv_dct4_1d_internal_c(c, stride << 1, min, max, tx64);
  104|       |
  105|  6.02M|    const int in1 = c[1 * stride], in3 = c[3 * stride];
  106|       |
  107|  6.02M|    int t4a, t5a, t6a, t7a;
  108|  6.02M|    if (tx64) {
  ------------------
  |  Branch (108:9): [True: 1.93M, False: 4.08M]
  ------------------
  109|  1.93M|        t4a = (in1 *   799 + 2048) >> 12;
  110|  1.93M|        t5a = (in3 * -2276 + 2048) >> 12;
  111|  1.93M|        t6a = (in3 *  3406 + 2048) >> 12;
  112|  1.93M|        t7a = (in1 *  4017 + 2048) >> 12;
  113|  4.08M|    } else {
  114|  4.08M|        const int in5 = c[5 * stride], in7 = c[7 * stride];
  115|       |
  116|  4.08M|        t4a = ((in1 *   799         - in7 * (4017 - 4096) + 2048) >> 12) - in7;
  117|  4.08M|        t5a =  (in5 *  1703         - in3 *  1138         + 1024) >> 11;
  118|  4.08M|        t6a =  (in5 *  1138         + in3 *  1703         + 1024) >> 11;
  119|  4.08M|        t7a = ((in1 * (4017 - 4096) + in7 *  799          + 2048) >> 12) + in1;
  120|  4.08M|    }
  121|       |
  122|  6.02M|    const int t4  = CLIP(t4a + t5a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  123|  6.02M|              t5a = CLIP(t4a - t5a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  124|  6.02M|    const int t7  = CLIP(t7a + t6a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  125|  6.02M|              t6a = CLIP(t7a - t6a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  126|       |
  127|  6.02M|    const int t5  = ((t6a - t5a) * 181 + 128) >> 8;
  128|  6.02M|    const int t6  = ((t6a + t5a) * 181 + 128) >> 8;
  129|       |
  130|  6.02M|    const int t0 = c[0 * stride];
  131|  6.02M|    const int t1 = c[2 * stride];
  132|  6.02M|    const int t2 = c[4 * stride];
  133|  6.02M|    const int t3 = c[6 * stride];
  134|       |
  135|  6.02M|    c[0 * stride] = CLIP(t0 + t7);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  136|  6.02M|    c[1 * stride] = CLIP(t1 + t6);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  137|  6.02M|    c[2 * stride] = CLIP(t2 + t5);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  138|  6.02M|    c[3 * stride] = CLIP(t3 + t4);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  139|  6.02M|    c[4 * stride] = CLIP(t3 - t4);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  140|  6.02M|    c[5 * stride] = CLIP(t2 - t5);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  141|  6.02M|    c[6 * stride] = CLIP(t1 - t6);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  142|  6.02M|    c[7 * stride] = CLIP(t0 - t7);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  143|  6.02M|}
itx_1d.c:inv_dct16_1d_c:
  242|  1.26M|{
  243|  1.26M|    inv_dct16_1d_internal_c(c, stride, min, max, 0);
  244|  1.26M|}
itx_1d.c:inv_dct16_1d_internal_c:
  154|  6.02M|{
  155|  6.02M|    assert(stride > 0);
  ------------------
  |  Branch (155:5): [True: 6.02M, False: 0]
  ------------------
  156|  6.02M|    inv_dct8_1d_internal_c(c, stride << 1, min, max, tx64);
  157|       |
  158|  6.02M|    const int in1 = c[1 * stride], in3 = c[3 * stride];
  159|  6.02M|    const int in5 = c[5 * stride], in7 = c[7 * stride];
  160|       |
  161|  6.02M|    int t8a, t9a, t10a, t11a, t12a, t13a, t14a, t15a;
  162|  6.02M|    if (tx64) {
  ------------------
  |  Branch (162:9): [True: 1.93M, False: 4.08M]
  ------------------
  163|  1.93M|        t8a  = (in1 *   401 + 2048) >> 12;
  164|  1.93M|        t9a  = (in7 * -2598 + 2048) >> 12;
  165|  1.93M|        t10a = (in5 *  1931 + 2048) >> 12;
  166|  1.93M|        t11a = (in3 * -1189 + 2048) >> 12;
  167|  1.93M|        t12a = (in3 *  3920 + 2048) >> 12;
  168|  1.93M|        t13a = (in5 *  3612 + 2048) >> 12;
  169|  1.93M|        t14a = (in7 *  3166 + 2048) >> 12;
  170|  1.93M|        t15a = (in1 *  4076 + 2048) >> 12;
  171|  4.08M|    } else {
  172|  4.08M|        const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  173|  4.08M|        const int in13 = c[13 * stride], in15 = c[15 * stride];
  174|       |
  175|  4.08M|        t8a  = ((in1  *   401         - in15 * (4076 - 4096) + 2048) >> 12) - in15;
  176|  4.08M|        t9a  =  (in9  *  1583         - in7  *  1299         + 1024) >> 11;
  177|  4.08M|        t10a = ((in5  *  1931         - in11 * (3612 - 4096) + 2048) >> 12) - in11;
  178|  4.08M|        t11a = ((in13 * (3920 - 4096) - in3  *  1189         + 2048) >> 12) + in13;
  179|  4.08M|        t12a = ((in13 *  1189         + in3  * (3920 - 4096) + 2048) >> 12) + in3;
  180|  4.08M|        t13a = ((in5  * (3612 - 4096) + in11 *  1931         + 2048) >> 12) + in5;
  181|  4.08M|        t14a =  (in9  *  1299         + in7  *  1583         + 1024) >> 11;
  182|  4.08M|        t15a = ((in1  * (4076 - 4096) + in15 *   401         + 2048) >> 12) + in1;
  183|  4.08M|    }
  184|       |
  185|  6.02M|    int t8  = CLIP(t8a  + t9a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  186|  6.02M|    int t9  = CLIP(t8a  - t9a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  187|  6.02M|    int t10 = CLIP(t11a - t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  188|  6.02M|    int t11 = CLIP(t11a + t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  189|  6.02M|    int t12 = CLIP(t12a + t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  190|  6.02M|    int t13 = CLIP(t12a - t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  191|  6.02M|    int t14 = CLIP(t15a - t14a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  192|  6.02M|    int t15 = CLIP(t15a + t14a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  193|       |
  194|  6.02M|    t9a  = ((  t14 *  1567         - t9  * (3784 - 4096)  + 2048) >> 12) - t9;
  195|  6.02M|    t14a = ((  t14 * (3784 - 4096) + t9  *  1567          + 2048) >> 12) + t14;
  196|  6.02M|    t10a = ((-(t13 * (3784 - 4096) + t10 *  1567)         + 2048) >> 12) - t13;
  197|  6.02M|    t13a = ((  t13 *  1567         - t10 * (3784 - 4096)  + 2048) >> 12) - t10;
  198|       |
  199|  6.02M|    t8a  = CLIP(t8   + t11);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  200|  6.02M|    t9   = CLIP(t9a  + t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  201|  6.02M|    t10  = CLIP(t9a  - t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  202|  6.02M|    t11a = CLIP(t8   - t11);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  203|  6.02M|    t12a = CLIP(t15  - t12);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  204|  6.02M|    t13  = CLIP(t14a - t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  205|  6.02M|    t14  = CLIP(t14a + t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  206|  6.02M|    t15a = CLIP(t15  + t12);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  207|       |
  208|  6.02M|    t10a = ((t13  - t10)  * 181 + 128) >> 8;
  209|  6.02M|    t13a = ((t13  + t10)  * 181 + 128) >> 8;
  210|  6.02M|    t11  = ((t12a - t11a) * 181 + 128) >> 8;
  211|  6.02M|    t12  = ((t12a + t11a) * 181 + 128) >> 8;
  212|       |
  213|  6.02M|    const int t0 = c[ 0 * stride];
  214|  6.02M|    const int t1 = c[ 2 * stride];
  215|  6.02M|    const int t2 = c[ 4 * stride];
  216|  6.02M|    const int t3 = c[ 6 * stride];
  217|  6.02M|    const int t4 = c[ 8 * stride];
  218|  6.02M|    const int t5 = c[10 * stride];
  219|  6.02M|    const int t6 = c[12 * stride];
  220|  6.02M|    const int t7 = c[14 * stride];
  221|       |
  222|  6.02M|    c[ 0 * stride] = CLIP(t0 + t15a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  223|  6.02M|    c[ 1 * stride] = CLIP(t1 + t14);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  224|  6.02M|    c[ 2 * stride] = CLIP(t2 + t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  225|  6.02M|    c[ 3 * stride] = CLIP(t3 + t12);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  226|  6.02M|    c[ 4 * stride] = CLIP(t4 + t11);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  227|  6.02M|    c[ 5 * stride] = CLIP(t5 + t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  228|  6.02M|    c[ 6 * stride] = CLIP(t6 + t9);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  229|  6.02M|    c[ 7 * stride] = CLIP(t7 + t8a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  230|  6.02M|    c[ 8 * stride] = CLIP(t7 - t8a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  231|  6.02M|    c[ 9 * stride] = CLIP(t6 - t9);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  232|  6.02M|    c[10 * stride] = CLIP(t5 - t10a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  233|  6.02M|    c[11 * stride] = CLIP(t4 - t11);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  234|  6.02M|    c[12 * stride] = CLIP(t3 - t12);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  235|  6.02M|    c[13 * stride] = CLIP(t2 - t13a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  236|  6.02M|    c[14 * stride] = CLIP(t1 - t14);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  237|  6.02M|    c[15 * stride] = CLIP(t0 - t15a);
  ------------------
  |  |   37|  6.02M|#define CLIP(a) iclip(a, min, max)
  ------------------
  238|  6.02M|}
itx_1d.c:inv_dct32_1d_c:
  432|  2.82M|{
  433|  2.82M|    inv_dct32_1d_internal_c(c, stride, min, max, 0);
  434|  2.82M|}
itx_1d.c:inv_dct32_1d_internal_c:
  249|  4.75M|{
  250|  4.75M|    assert(stride > 0);
  ------------------
  |  Branch (250:5): [True: 4.75M, False: 0]
  ------------------
  251|  4.75M|    inv_dct16_1d_internal_c(c, stride << 1, min, max, tx64);
  252|       |
  253|  4.75M|    const int in1  = c[ 1 * stride], in3  = c[ 3 * stride];
  254|  4.75M|    const int in5  = c[ 5 * stride], in7  = c[ 7 * stride];
  255|  4.75M|    const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  256|  4.75M|    const int in13 = c[13 * stride], in15 = c[15 * stride];
  257|       |
  258|  4.75M|    int t16a, t17a, t18a, t19a, t20a, t21a, t22a, t23a;
  259|  4.75M|    int t24a, t25a, t26a, t27a, t28a, t29a, t30a, t31a;
  260|  4.75M|    if (tx64) {
  ------------------
  |  Branch (260:9): [True: 1.93M, False: 2.82M]
  ------------------
  261|  1.93M|        t16a = (in1  *   201 + 2048) >> 12;
  262|  1.93M|        t17a = (in15 * -2751 + 2048) >> 12;
  263|  1.93M|        t18a = (in9  *  1751 + 2048) >> 12;
  264|  1.93M|        t19a = (in7  * -1380 + 2048) >> 12;
  265|  1.93M|        t20a = (in5  *   995 + 2048) >> 12;
  266|  1.93M|        t21a = (in11 * -2106 + 2048) >> 12;
  267|  1.93M|        t22a = (in13 *  2440 + 2048) >> 12;
  268|  1.93M|        t23a = (in3  *  -601 + 2048) >> 12;
  269|  1.93M|        t24a = (in3  *  4052 + 2048) >> 12;
  270|  1.93M|        t25a = (in13 *  3290 + 2048) >> 12;
  271|  1.93M|        t26a = (in11 *  3513 + 2048) >> 12;
  272|  1.93M|        t27a = (in5  *  3973 + 2048) >> 12;
  273|  1.93M|        t28a = (in7  *  3857 + 2048) >> 12;
  274|  1.93M|        t29a = (in9  *  3703 + 2048) >> 12;
  275|  1.93M|        t30a = (in15 *  3035 + 2048) >> 12;
  276|  1.93M|        t31a = (in1  *  4091 + 2048) >> 12;
  277|  2.82M|    } else {
  278|  2.82M|        const int in17 = c[17 * stride], in19 = c[19 * stride];
  279|  2.82M|        const int in21 = c[21 * stride], in23 = c[23 * stride];
  280|  2.82M|        const int in25 = c[25 * stride], in27 = c[27 * stride];
  281|  2.82M|        const int in29 = c[29 * stride], in31 = c[31 * stride];
  282|       |
  283|  2.82M|        t16a = ((in1  *   201         - in31 * (4091 - 4096) + 2048) >> 12) - in31;
  284|  2.82M|        t17a = ((in17 * (3035 - 4096) - in15 *  2751         + 2048) >> 12) + in17;
  285|  2.82M|        t18a = ((in9  *  1751         - in23 * (3703 - 4096) + 2048) >> 12) - in23;
  286|  2.82M|        t19a = ((in25 * (3857 - 4096) - in7  *  1380         + 2048) >> 12) + in25;
  287|  2.82M|        t20a = ((in5  *   995         - in27 * (3973 - 4096) + 2048) >> 12) - in27;
  288|  2.82M|        t21a = ((in21 * (3513 - 4096) - in11 *  2106         + 2048) >> 12) + in21;
  289|  2.82M|        t22a =  (in13 *  1220         - in19 *  1645         + 1024) >> 11;
  290|  2.82M|        t23a = ((in29 * (4052 - 4096) - in3  *   601         + 2048) >> 12) + in29;
  291|  2.82M|        t24a = ((in29 *   601         + in3  * (4052 - 4096) + 2048) >> 12) + in3;
  292|  2.82M|        t25a =  (in13 *  1645         + in19 *  1220         + 1024) >> 11;
  293|  2.82M|        t26a = ((in21 *  2106         + in11 * (3513 - 4096) + 2048) >> 12) + in11;
  294|  2.82M|        t27a = ((in5  * (3973 - 4096) + in27 *   995         + 2048) >> 12) + in5;
  295|  2.82M|        t28a = ((in25 *  1380         + in7  * (3857 - 4096) + 2048) >> 12) + in7;
  296|  2.82M|        t29a = ((in9  * (3703 - 4096) + in23 *  1751         + 2048) >> 12) + in9;
  297|  2.82M|        t30a = ((in17 *  2751         + in15 * (3035 - 4096) + 2048) >> 12) + in15;
  298|  2.82M|        t31a = ((in1  * (4091 - 4096) + in31 *   201         + 2048) >> 12) + in1;
  299|  2.82M|    }
  300|       |
  301|  4.75M|    int t16 = CLIP(t16a + t17a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  302|  4.75M|    int t17 = CLIP(t16a - t17a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  303|  4.75M|    int t18 = CLIP(t19a - t18a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  304|  4.75M|    int t19 = CLIP(t19a + t18a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  305|  4.75M|    int t20 = CLIP(t20a + t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  306|  4.75M|    int t21 = CLIP(t20a - t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  307|  4.75M|    int t22 = CLIP(t23a - t22a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  308|  4.75M|    int t23 = CLIP(t23a + t22a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  309|  4.75M|    int t24 = CLIP(t24a + t25a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  310|  4.75M|    int t25 = CLIP(t24a - t25a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  311|  4.75M|    int t26 = CLIP(t27a - t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  312|  4.75M|    int t27 = CLIP(t27a + t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  313|  4.75M|    int t28 = CLIP(t28a + t29a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  314|  4.75M|    int t29 = CLIP(t28a - t29a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  315|  4.75M|    int t30 = CLIP(t31a - t30a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  316|  4.75M|    int t31 = CLIP(t31a + t30a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  317|       |
  318|  4.75M|    t17a = ((  t30 *   799         - t17 * (4017 - 4096)  + 2048) >> 12) - t17;
  319|  4.75M|    t30a = ((  t30 * (4017 - 4096) + t17 *   799          + 2048) >> 12) + t30;
  320|  4.75M|    t18a = ((-(t29 * (4017 - 4096) + t18 *   799)         + 2048) >> 12) - t29;
  321|  4.75M|    t29a = ((  t29 *   799         - t18 * (4017 - 4096)  + 2048) >> 12) - t18;
  322|  4.75M|    t21a =  (  t26 *  1703         - t21 *  1138          + 1024) >> 11;
  323|  4.75M|    t26a =  (  t26 *  1138         + t21 *  1703          + 1024) >> 11;
  324|  4.75M|    t22a =  (-(t25 *  1138         + t22 *  1703        ) + 1024) >> 11;
  325|  4.75M|    t25a =  (  t25 *  1703         - t22 *  1138          + 1024) >> 11;
  326|       |
  327|  4.75M|    t16a = CLIP(t16  + t19);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  328|  4.75M|    t17  = CLIP(t17a + t18a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  329|  4.75M|    t18  = CLIP(t17a - t18a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  330|  4.75M|    t19a = CLIP(t16  - t19);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  331|  4.75M|    t20a = CLIP(t23  - t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  332|  4.75M|    t21  = CLIP(t22a - t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  333|  4.75M|    t22  = CLIP(t22a + t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  334|  4.75M|    t23a = CLIP(t23  + t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  335|  4.75M|    t24a = CLIP(t24  + t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  336|  4.75M|    t25  = CLIP(t25a + t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  337|  4.75M|    t26  = CLIP(t25a - t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  338|  4.75M|    t27a = CLIP(t24  - t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  339|  4.75M|    t28a = CLIP(t31  - t28);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  340|  4.75M|    t29  = CLIP(t30a - t29a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  341|  4.75M|    t30  = CLIP(t30a + t29a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  342|  4.75M|    t31a = CLIP(t31  + t28);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  343|       |
  344|  4.75M|    t18a = ((  t29  *  1567         - t18  * (3784 - 4096)  + 2048) >> 12) - t18;
  345|  4.75M|    t29a = ((  t29  * (3784 - 4096) + t18  *  1567          + 2048) >> 12) + t29;
  346|  4.75M|    t19  = ((  t28a *  1567         - t19a * (3784 - 4096)  + 2048) >> 12) - t19a;
  347|  4.75M|    t28  = ((  t28a * (3784 - 4096) + t19a *  1567          + 2048) >> 12) + t28a;
  348|  4.75M|    t20  = ((-(t27a * (3784 - 4096) + t20a *  1567)         + 2048) >> 12) - t27a;
  349|  4.75M|    t27  = ((  t27a *  1567         - t20a * (3784 - 4096)  + 2048) >> 12) - t20a;
  350|  4.75M|    t21a = ((-(t26  * (3784 - 4096) + t21  *  1567)         + 2048) >> 12) - t26;
  351|  4.75M|    t26a = ((  t26  *  1567         - t21  * (3784 - 4096)  + 2048) >> 12) - t21;
  352|       |
  353|  4.75M|    t16  = CLIP(t16a + t23a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  354|  4.75M|    t17a = CLIP(t17  + t22);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  355|  4.75M|    t18  = CLIP(t18a + t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  356|  4.75M|    t19a = CLIP(t19  + t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  357|  4.75M|    t20a = CLIP(t19  - t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  358|  4.75M|    t21  = CLIP(t18a - t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  359|  4.75M|    t22a = CLIP(t17  - t22);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  360|  4.75M|    t23  = CLIP(t16a - t23a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  361|  4.75M|    t24  = CLIP(t31a - t24a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  362|  4.75M|    t25a = CLIP(t30  - t25);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  363|  4.75M|    t26  = CLIP(t29a - t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  364|  4.75M|    t27a = CLIP(t28  - t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  365|  4.75M|    t28a = CLIP(t28  + t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  366|  4.75M|    t29  = CLIP(t29a + t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  367|  4.75M|    t30a = CLIP(t30  + t25);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  368|  4.75M|    t31  = CLIP(t31a + t24a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  369|       |
  370|  4.75M|    t20  = ((t27a - t20a) * 181 + 128) >> 8;
  371|  4.75M|    t27  = ((t27a + t20a) * 181 + 128) >> 8;
  372|  4.75M|    t21a = ((t26  - t21 ) * 181 + 128) >> 8;
  373|  4.75M|    t26a = ((t26  + t21 ) * 181 + 128) >> 8;
  374|  4.75M|    t22  = ((t25a - t22a) * 181 + 128) >> 8;
  375|  4.75M|    t25  = ((t25a + t22a) * 181 + 128) >> 8;
  376|  4.75M|    t23a = ((t24  - t23 ) * 181 + 128) >> 8;
  377|  4.75M|    t24a = ((t24  + t23 ) * 181 + 128) >> 8;
  378|       |
  379|  4.75M|    const int t0  = c[ 0 * stride];
  380|  4.75M|    const int t1  = c[ 2 * stride];
  381|  4.75M|    const int t2  = c[ 4 * stride];
  382|  4.75M|    const int t3  = c[ 6 * stride];
  383|  4.75M|    const int t4  = c[ 8 * stride];
  384|  4.75M|    const int t5  = c[10 * stride];
  385|  4.75M|    const int t6  = c[12 * stride];
  386|  4.75M|    const int t7  = c[14 * stride];
  387|  4.75M|    const int t8  = c[16 * stride];
  388|  4.75M|    const int t9  = c[18 * stride];
  389|  4.75M|    const int t10 = c[20 * stride];
  390|  4.75M|    const int t11 = c[22 * stride];
  391|  4.75M|    const int t12 = c[24 * stride];
  392|  4.75M|    const int t13 = c[26 * stride];
  393|  4.75M|    const int t14 = c[28 * stride];
  394|  4.75M|    const int t15 = c[30 * stride];
  395|       |
  396|  4.75M|    c[ 0 * stride] = CLIP(t0  + t31);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  397|  4.75M|    c[ 1 * stride] = CLIP(t1  + t30a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  398|  4.75M|    c[ 2 * stride] = CLIP(t2  + t29);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  399|  4.75M|    c[ 3 * stride] = CLIP(t3  + t28a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  400|  4.75M|    c[ 4 * stride] = CLIP(t4  + t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  401|  4.75M|    c[ 5 * stride] = CLIP(t5  + t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  402|  4.75M|    c[ 6 * stride] = CLIP(t6  + t25);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  403|  4.75M|    c[ 7 * stride] = CLIP(t7  + t24a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  404|  4.75M|    c[ 8 * stride] = CLIP(t8  + t23a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  405|  4.75M|    c[ 9 * stride] = CLIP(t9  + t22);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  406|  4.75M|    c[10 * stride] = CLIP(t10 + t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  407|  4.75M|    c[11 * stride] = CLIP(t11 + t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  408|  4.75M|    c[12 * stride] = CLIP(t12 + t19a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  409|  4.75M|    c[13 * stride] = CLIP(t13 + t18);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  410|  4.75M|    c[14 * stride] = CLIP(t14 + t17a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  411|  4.75M|    c[15 * stride] = CLIP(t15 + t16);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  412|  4.75M|    c[16 * stride] = CLIP(t15 - t16);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  413|  4.75M|    c[17 * stride] = CLIP(t14 - t17a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  414|  4.75M|    c[18 * stride] = CLIP(t13 - t18);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  415|  4.75M|    c[19 * stride] = CLIP(t12 - t19a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  416|  4.75M|    c[20 * stride] = CLIP(t11 - t20);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  417|  4.75M|    c[21 * stride] = CLIP(t10 - t21a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  418|  4.75M|    c[22 * stride] = CLIP(t9  - t22);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  419|  4.75M|    c[23 * stride] = CLIP(t8  - t23a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  420|  4.75M|    c[24 * stride] = CLIP(t7  - t24a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  421|  4.75M|    c[25 * stride] = CLIP(t6  - t25);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  422|  4.75M|    c[26 * stride] = CLIP(t5  - t26a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  423|  4.75M|    c[27 * stride] = CLIP(t4  - t27);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  424|  4.75M|    c[28 * stride] = CLIP(t3  - t28a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  425|  4.75M|    c[29 * stride] = CLIP(t2  - t29);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  426|  4.75M|    c[30 * stride] = CLIP(t1  - t30a);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  427|  4.75M|    c[31 * stride] = CLIP(t0  - t31);
  ------------------
  |  |   37|  4.75M|#define CLIP(a) iclip(a, min, max)
  ------------------
  428|  4.75M|}
itx_1d.c:inv_dct64_1d_c:
  438|  1.93M|{
  439|  1.93M|    assert(stride > 0);
  ------------------
  |  Branch (439:5): [True: 1.93M, False: 0]
  ------------------
  440|  1.93M|    inv_dct32_1d_internal_c(c, stride << 1, min, max, 1);
  441|       |
  442|  1.93M|    const int in1  = c[ 1 * stride], in3  = c[ 3 * stride];
  443|  1.93M|    const int in5  = c[ 5 * stride], in7  = c[ 7 * stride];
  444|  1.93M|    const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  445|  1.93M|    const int in13 = c[13 * stride], in15 = c[15 * stride];
  446|  1.93M|    const int in17 = c[17 * stride], in19 = c[19 * stride];
  447|  1.93M|    const int in21 = c[21 * stride], in23 = c[23 * stride];
  448|  1.93M|    const int in25 = c[25 * stride], in27 = c[27 * stride];
  449|  1.93M|    const int in29 = c[29 * stride], in31 = c[31 * stride];
  450|       |
  451|  1.93M|    int t32a = (in1  *   101 + 2048) >> 12;
  452|  1.93M|    int t33a = (in31 * -2824 + 2048) >> 12;
  453|  1.93M|    int t34a = (in17 *  1660 + 2048) >> 12;
  454|  1.93M|    int t35a = (in15 * -1474 + 2048) >> 12;
  455|  1.93M|    int t36a = (in9  *   897 + 2048) >> 12;
  456|  1.93M|    int t37a = (in23 * -2191 + 2048) >> 12;
  457|  1.93M|    int t38a = (in25 *  2359 + 2048) >> 12;
  458|  1.93M|    int t39a = (in7  *  -700 + 2048) >> 12;
  459|  1.93M|    int t40a = (in5  *   501 + 2048) >> 12;
  460|  1.93M|    int t41a = (in27 * -2520 + 2048) >> 12;
  461|  1.93M|    int t42a = (in21 *  2019 + 2048) >> 12;
  462|  1.93M|    int t43a = (in11 * -1092 + 2048) >> 12;
  463|  1.93M|    int t44a = (in13 *  1285 + 2048) >> 12;
  464|  1.93M|    int t45a = (in19 * -1842 + 2048) >> 12;
  465|  1.93M|    int t46a = (in29 *  2675 + 2048) >> 12;
  466|  1.93M|    int t47a = (in3  *  -301 + 2048) >> 12;
  467|  1.93M|    int t48a = (in3  *  4085 + 2048) >> 12;
  468|  1.93M|    int t49a = (in29 *  3102 + 2048) >> 12;
  469|  1.93M|    int t50a = (in19 *  3659 + 2048) >> 12;
  470|  1.93M|    int t51a = (in13 *  3889 + 2048) >> 12;
  471|  1.93M|    int t52a = (in11 *  3948 + 2048) >> 12;
  472|  1.93M|    int t53a = (in21 *  3564 + 2048) >> 12;
  473|  1.93M|    int t54a = (in27 *  3229 + 2048) >> 12;
  474|  1.93M|    int t55a = (in5  *  4065 + 2048) >> 12;
  475|  1.93M|    int t56a = (in7  *  4036 + 2048) >> 12;
  476|  1.93M|    int t57a = (in25 *  3349 + 2048) >> 12;
  477|  1.93M|    int t58a = (in23 *  3461 + 2048) >> 12;
  478|  1.93M|    int t59a = (in9  *  3996 + 2048) >> 12;
  479|  1.93M|    int t60a = (in15 *  3822 + 2048) >> 12;
  480|  1.93M|    int t61a = (in17 *  3745 + 2048) >> 12;
  481|  1.93M|    int t62a = (in31 *  2967 + 2048) >> 12;
  482|  1.93M|    int t63a = (in1  *  4095 + 2048) >> 12;
  483|       |
  484|  1.93M|    int t32 = CLIP(t32a + t33a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  485|  1.93M|    int t33 = CLIP(t32a - t33a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  486|  1.93M|    int t34 = CLIP(t35a - t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  487|  1.93M|    int t35 = CLIP(t35a + t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  488|  1.93M|    int t36 = CLIP(t36a + t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  489|  1.93M|    int t37 = CLIP(t36a - t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  490|  1.93M|    int t38 = CLIP(t39a - t38a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  491|  1.93M|    int t39 = CLIP(t39a + t38a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  492|  1.93M|    int t40 = CLIP(t40a + t41a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  493|  1.93M|    int t41 = CLIP(t40a - t41a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  494|  1.93M|    int t42 = CLIP(t43a - t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  495|  1.93M|    int t43 = CLIP(t43a + t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  496|  1.93M|    int t44 = CLIP(t44a + t45a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  497|  1.93M|    int t45 = CLIP(t44a - t45a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  498|  1.93M|    int t46 = CLIP(t47a - t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  499|  1.93M|    int t47 = CLIP(t47a + t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  500|  1.93M|    int t48 = CLIP(t48a + t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  501|  1.93M|    int t49 = CLIP(t48a - t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  502|  1.93M|    int t50 = CLIP(t51a - t50a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  503|  1.93M|    int t51 = CLIP(t51a + t50a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  504|  1.93M|    int t52 = CLIP(t52a + t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  505|  1.93M|    int t53 = CLIP(t52a - t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  506|  1.93M|    int t54 = CLIP(t55a - t54a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  507|  1.93M|    int t55 = CLIP(t55a + t54a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  508|  1.93M|    int t56 = CLIP(t56a + t57a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  509|  1.93M|    int t57 = CLIP(t56a - t57a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  510|  1.93M|    int t58 = CLIP(t59a - t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  511|  1.93M|    int t59 = CLIP(t59a + t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  512|  1.93M|    int t60 = CLIP(t60a + t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  513|  1.93M|    int t61 = CLIP(t60a - t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  514|  1.93M|    int t62 = CLIP(t63a - t62a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  515|  1.93M|    int t63 = CLIP(t63a + t62a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  516|       |
  517|  1.93M|    t33a = ((t33 * (4096 - 4076) + t62 *   401         + 2048) >> 12) - t33;
  518|  1.93M|    t34a = ((t34 *  -401         + t61 * (4096 - 4076) + 2048) >> 12) - t61;
  519|  1.93M|    t37a =  (t37 * -1299         + t58 *  1583         + 1024) >> 11;
  520|  1.93M|    t38a =  (t38 * -1583         + t57 * -1299         + 1024) >> 11;
  521|  1.93M|    t41a = ((t41 * (4096 - 3612) + t54 *  1931         + 2048) >> 12) - t41;
  522|  1.93M|    t42a = ((t42 * -1931         + t53 * (4096 - 3612) + 2048) >> 12) - t53;
  523|  1.93M|    t45a = ((t45 * -1189         + t50 * (3920 - 4096) + 2048) >> 12) + t50;
  524|  1.93M|    t46a = ((t46 * (4096 - 3920) + t49 * -1189         + 2048) >> 12) - t46;
  525|  1.93M|    t49a = ((t46 * -1189         + t49 * (3920 - 4096) + 2048) >> 12) + t49;
  526|  1.93M|    t50a = ((t45 * (3920 - 4096) + t50 *  1189         + 2048) >> 12) + t45;
  527|  1.93M|    t53a = ((t42 * (4096 - 3612) + t53 *  1931         + 2048) >> 12) - t42;
  528|  1.93M|    t54a = ((t41 *  1931         + t54 * (3612 - 4096) + 2048) >> 12) + t54;
  529|  1.93M|    t57a =  (t38 * -1299         + t57 *  1583         + 1024) >> 11;
  530|  1.93M|    t58a =  (t37 *  1583         + t58 *  1299         + 1024) >> 11;
  531|  1.93M|    t61a = ((t34 * (4096 - 4076) + t61 *   401         + 2048) >> 12) - t34;
  532|  1.93M|    t62a = ((t33 *   401         + t62 * (4076 - 4096) + 2048) >> 12) + t62;
  533|       |
  534|  1.93M|    t32a = CLIP(t32  + t35);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  535|  1.93M|    t33  = CLIP(t33a + t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  536|  1.93M|    t34  = CLIP(t33a - t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  537|  1.93M|    t35a = CLIP(t32  - t35);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  538|  1.93M|    t36a = CLIP(t39  - t36);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  539|  1.93M|    t37  = CLIP(t38a - t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  540|  1.93M|    t38  = CLIP(t38a + t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  541|  1.93M|    t39a = CLIP(t39  + t36);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  542|  1.93M|    t40a = CLIP(t40  + t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  543|  1.93M|    t41  = CLIP(t41a + t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  544|  1.93M|    t42  = CLIP(t41a - t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  545|  1.93M|    t43a = CLIP(t40  - t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  546|  1.93M|    t44a = CLIP(t47  - t44);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  547|  1.93M|    t45  = CLIP(t46a - t45a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  548|  1.93M|    t46  = CLIP(t46a + t45a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  549|  1.93M|    t47a = CLIP(t47  + t44);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  550|  1.93M|    t48a = CLIP(t48  + t51);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  551|  1.93M|    t49  = CLIP(t49a + t50a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  552|  1.93M|    t50  = CLIP(t49a - t50a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  553|  1.93M|    t51a = CLIP(t48  - t51);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  554|  1.93M|    t52a = CLIP(t55  - t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  555|  1.93M|    t53  = CLIP(t54a - t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  556|  1.93M|    t54  = CLIP(t54a + t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  557|  1.93M|    t55a = CLIP(t55  + t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  558|  1.93M|    t56a = CLIP(t56  + t59);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  559|  1.93M|    t57  = CLIP(t57a + t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  560|  1.93M|    t58  = CLIP(t57a - t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  561|  1.93M|    t59a = CLIP(t56  - t59);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  562|  1.93M|    t60a = CLIP(t63  - t60);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  563|  1.93M|    t61  = CLIP(t62a - t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  564|  1.93M|    t62  = CLIP(t62a + t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  565|  1.93M|    t63a = CLIP(t63  + t60);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  566|       |
  567|  1.93M|    t34a = ((t34  * (4096 - 4017) + t61  *   799         + 2048) >> 12) - t34;
  568|  1.93M|    t35  = ((t35a * (4096 - 4017) + t60a *   799         + 2048) >> 12) - t35a;
  569|  1.93M|    t36  = ((t36a *  -799         + t59a * (4096 - 4017) + 2048) >> 12) - t59a;
  570|  1.93M|    t37a = ((t37  *  -799         + t58  * (4096 - 4017) + 2048) >> 12) - t58;
  571|  1.93M|    t42a =  (t42  * -1138         + t53  *  1703         + 1024) >> 11;
  572|  1.93M|    t43  =  (t43a * -1138         + t52a *  1703         + 1024) >> 11;
  573|  1.93M|    t44  =  (t44a * -1703         + t51a * -1138         + 1024) >> 11;
  574|  1.93M|    t45a =  (t45  * -1703         + t50  * -1138         + 1024) >> 11;
  575|  1.93M|    t50a =  (t45  * -1138         + t50  *  1703         + 1024) >> 11;
  576|  1.93M|    t51  =  (t44a * -1138         + t51a *  1703         + 1024) >> 11;
  577|  1.93M|    t52  =  (t43a *  1703         + t52a *  1138         + 1024) >> 11;
  578|  1.93M|    t53a =  (t42  *  1703         + t53  *  1138         + 1024) >> 11;
  579|  1.93M|    t58a = ((t37  * (4096 - 4017) + t58  *   799         + 2048) >> 12) - t37;
  580|  1.93M|    t59  = ((t36a * (4096 - 4017) + t59a *   799         + 2048) >> 12) - t36a;
  581|  1.93M|    t60  = ((t35a *   799         + t60a * (4017 - 4096) + 2048) >> 12) + t60a;
  582|  1.93M|    t61a = ((t34  *   799         + t61  * (4017 - 4096) + 2048) >> 12) + t61;
  583|       |
  584|  1.93M|    t32  = CLIP(t32a + t39a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  585|  1.93M|    t33a = CLIP(t33  + t38);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  586|  1.93M|    t34  = CLIP(t34a + t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  587|  1.93M|    t35a = CLIP(t35  + t36);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  588|  1.93M|    t36a = CLIP(t35  - t36);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  589|  1.93M|    t37  = CLIP(t34a - t37a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  590|  1.93M|    t38a = CLIP(t33  - t38);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  591|  1.93M|    t39  = CLIP(t32a - t39a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  592|  1.93M|    t40  = CLIP(t47a - t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  593|  1.93M|    t41a = CLIP(t46  - t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  594|  1.93M|    t42  = CLIP(t45a - t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  595|  1.93M|    t43a = CLIP(t44  - t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  596|  1.93M|    t44a = CLIP(t44  + t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  597|  1.93M|    t45  = CLIP(t45a + t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  598|  1.93M|    t46a = CLIP(t46  + t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  599|  1.93M|    t47  = CLIP(t47a + t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  600|  1.93M|    t48  = CLIP(t48a + t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  601|  1.93M|    t49a = CLIP(t49  + t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  602|  1.93M|    t50  = CLIP(t50a + t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  603|  1.93M|    t51a = CLIP(t51  + t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  604|  1.93M|    t52a = CLIP(t51  - t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  605|  1.93M|    t53  = CLIP(t50a - t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  606|  1.93M|    t54a = CLIP(t49  - t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  607|  1.93M|    t55  = CLIP(t48a - t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  608|  1.93M|    t56  = CLIP(t63a - t56a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  609|  1.93M|    t57a = CLIP(t62  - t57);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  610|  1.93M|    t58  = CLIP(t61a - t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  611|  1.93M|    t59a = CLIP(t60  - t59);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  612|  1.93M|    t60a = CLIP(t60  + t59);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  613|  1.93M|    t61  = CLIP(t61a + t58a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  614|  1.93M|    t62a = CLIP(t62  + t57);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  615|  1.93M|    t63  = CLIP(t63a + t56a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  616|       |
  617|  1.93M|    t36  = ((t36a * (4096 - 3784) + t59a *  1567         + 2048) >> 12) - t36a;
  618|  1.93M|    t37a = ((t37  * (4096 - 3784) + t58  *  1567         + 2048) >> 12) - t37;
  619|  1.93M|    t38  = ((t38a * (4096 - 3784) + t57a *  1567         + 2048) >> 12) - t38a;
  620|  1.93M|    t39a = ((t39  * (4096 - 3784) + t56  *  1567         + 2048) >> 12) - t39;
  621|  1.93M|    t40a = ((t40  * -1567         + t55  * (4096 - 3784) + 2048) >> 12) - t55;
  622|  1.93M|    t41  = ((t41a * -1567         + t54a * (4096 - 3784) + 2048) >> 12) - t54a;
  623|  1.93M|    t42a = ((t42  * -1567         + t53  * (4096 - 3784) + 2048) >> 12) - t53;
  624|  1.93M|    t43  = ((t43a * -1567         + t52a * (4096 - 3784) + 2048) >> 12) - t52a;
  625|  1.93M|    t52  = ((t43a * (4096 - 3784) + t52a *  1567         + 2048) >> 12) - t43a;
  626|  1.93M|    t53a = ((t42  * (4096 - 3784) + t53  *  1567         + 2048) >> 12) - t42;
  627|  1.93M|    t54  = ((t41a * (4096 - 3784) + t54a *  1567         + 2048) >> 12) - t41a;
  628|  1.93M|    t55a = ((t40  * (4096 - 3784) + t55  *  1567         + 2048) >> 12) - t40;
  629|  1.93M|    t56a = ((t39  *  1567         + t56  * (3784 - 4096) + 2048) >> 12) + t56;
  630|  1.93M|    t57  = ((t38a *  1567         + t57a * (3784 - 4096) + 2048) >> 12) + t57a;
  631|  1.93M|    t58a = ((t37  *  1567         + t58  * (3784 - 4096) + 2048) >> 12) + t58;
  632|  1.93M|    t59  = ((t36a *  1567         + t59a * (3784 - 4096) + 2048) >> 12) + t59a;
  633|       |
  634|  1.93M|    t32a = CLIP(t32  + t47);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  635|  1.93M|    t33  = CLIP(t33a + t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  636|  1.93M|    t34a = CLIP(t34  + t45);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  637|  1.93M|    t35  = CLIP(t35a + t44a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  638|  1.93M|    t36a = CLIP(t36  + t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  639|  1.93M|    t37  = CLIP(t37a + t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  640|  1.93M|    t38a = CLIP(t38  + t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  641|  1.93M|    t39  = CLIP(t39a + t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  642|  1.93M|    t40  = CLIP(t39a - t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  643|  1.93M|    t41a = CLIP(t38  - t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  644|  1.93M|    t42  = CLIP(t37a - t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  645|  1.93M|    t43a = CLIP(t36  - t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  646|  1.93M|    t44  = CLIP(t35a - t44a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  647|  1.93M|    t45a = CLIP(t34  - t45);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  648|  1.93M|    t46  = CLIP(t33a - t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  649|  1.93M|    t47a = CLIP(t32  - t47);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  650|  1.93M|    t48a = CLIP(t63  - t48);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  651|  1.93M|    t49  = CLIP(t62a - t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  652|  1.93M|    t50a = CLIP(t61  - t50);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  653|  1.93M|    t51  = CLIP(t60a - t51a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  654|  1.93M|    t52a = CLIP(t59  - t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  655|  1.93M|    t53  = CLIP(t58a - t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  656|  1.93M|    t54a = CLIP(t57  - t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  657|  1.93M|    t55  = CLIP(t56a - t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  658|  1.93M|    t56  = CLIP(t56a + t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  659|  1.93M|    t57a = CLIP(t57  + t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  660|  1.93M|    t58  = CLIP(t58a + t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  661|  1.93M|    t59a = CLIP(t59  + t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  662|  1.93M|    t60  = CLIP(t60a + t51a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  663|  1.93M|    t61a = CLIP(t61  + t50);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  664|  1.93M|    t62  = CLIP(t62a + t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  665|  1.93M|    t63a = CLIP(t63  + t48);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  666|       |
  667|  1.93M|    t40a = ((t55  - t40 ) * 181 + 128) >> 8;
  668|  1.93M|    t41  = ((t54a - t41a) * 181 + 128) >> 8;
  669|  1.93M|    t42a = ((t53  - t42 ) * 181 + 128) >> 8;
  670|  1.93M|    t43  = ((t52a - t43a) * 181 + 128) >> 8;
  671|  1.93M|    t44a = ((t51  - t44 ) * 181 + 128) >> 8;
  672|  1.93M|    t45  = ((t50a - t45a) * 181 + 128) >> 8;
  673|  1.93M|    t46a = ((t49  - t46 ) * 181 + 128) >> 8;
  674|  1.93M|    t47  = ((t48a - t47a) * 181 + 128) >> 8;
  675|  1.93M|    t48  = ((t47a + t48a) * 181 + 128) >> 8;
  676|  1.93M|    t49a = ((t46  + t49 ) * 181 + 128) >> 8;
  677|  1.93M|    t50  = ((t45a + t50a) * 181 + 128) >> 8;
  678|  1.93M|    t51a = ((t44  + t51 ) * 181 + 128) >> 8;
  679|  1.93M|    t52  = ((t43a + t52a) * 181 + 128) >> 8;
  680|  1.93M|    t53a = ((t42  + t53 ) * 181 + 128) >> 8;
  681|  1.93M|    t54  = ((t41a + t54a) * 181 + 128) >> 8;
  682|  1.93M|    t55a = ((t40  + t55 ) * 181 + 128) >> 8;
  683|       |
  684|  1.93M|    const int t0  = c[ 0 * stride];
  685|  1.93M|    const int t1  = c[ 2 * stride];
  686|  1.93M|    const int t2  = c[ 4 * stride];
  687|  1.93M|    const int t3  = c[ 6 * stride];
  688|  1.93M|    const int t4  = c[ 8 * stride];
  689|  1.93M|    const int t5  = c[10 * stride];
  690|  1.93M|    const int t6  = c[12 * stride];
  691|  1.93M|    const int t7  = c[14 * stride];
  692|  1.93M|    const int t8  = c[16 * stride];
  693|  1.93M|    const int t9  = c[18 * stride];
  694|  1.93M|    const int t10 = c[20 * stride];
  695|  1.93M|    const int t11 = c[22 * stride];
  696|  1.93M|    const int t12 = c[24 * stride];
  697|  1.93M|    const int t13 = c[26 * stride];
  698|  1.93M|    const int t14 = c[28 * stride];
  699|  1.93M|    const int t15 = c[30 * stride];
  700|  1.93M|    const int t16 = c[32 * stride];
  701|  1.93M|    const int t17 = c[34 * stride];
  702|  1.93M|    const int t18 = c[36 * stride];
  703|  1.93M|    const int t19 = c[38 * stride];
  704|  1.93M|    const int t20 = c[40 * stride];
  705|  1.93M|    const int t21 = c[42 * stride];
  706|  1.93M|    const int t22 = c[44 * stride];
  707|  1.93M|    const int t23 = c[46 * stride];
  708|  1.93M|    const int t24 = c[48 * stride];
  709|  1.93M|    const int t25 = c[50 * stride];
  710|  1.93M|    const int t26 = c[52 * stride];
  711|  1.93M|    const int t27 = c[54 * stride];
  712|  1.93M|    const int t28 = c[56 * stride];
  713|  1.93M|    const int t29 = c[58 * stride];
  714|  1.93M|    const int t30 = c[60 * stride];
  715|  1.93M|    const int t31 = c[62 * stride];
  716|       |
  717|  1.93M|    c[ 0 * stride] = CLIP(t0  + t63a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  718|  1.93M|    c[ 1 * stride] = CLIP(t1  + t62);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  719|  1.93M|    c[ 2 * stride] = CLIP(t2  + t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  720|  1.93M|    c[ 3 * stride] = CLIP(t3  + t60);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  721|  1.93M|    c[ 4 * stride] = CLIP(t4  + t59a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  722|  1.93M|    c[ 5 * stride] = CLIP(t5  + t58);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  723|  1.93M|    c[ 6 * stride] = CLIP(t6  + t57a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  724|  1.93M|    c[ 7 * stride] = CLIP(t7  + t56);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  725|  1.93M|    c[ 8 * stride] = CLIP(t8  + t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  726|  1.93M|    c[ 9 * stride] = CLIP(t9  + t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  727|  1.93M|    c[10 * stride] = CLIP(t10 + t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  728|  1.93M|    c[11 * stride] = CLIP(t11 + t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  729|  1.93M|    c[12 * stride] = CLIP(t12 + t51a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  730|  1.93M|    c[13 * stride] = CLIP(t13 + t50);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  731|  1.93M|    c[14 * stride] = CLIP(t14 + t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  732|  1.93M|    c[15 * stride] = CLIP(t15 + t48);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  733|  1.93M|    c[16 * stride] = CLIP(t16 + t47);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  734|  1.93M|    c[17 * stride] = CLIP(t17 + t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  735|  1.93M|    c[18 * stride] = CLIP(t18 + t45);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  736|  1.93M|    c[19 * stride] = CLIP(t19 + t44a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  737|  1.93M|    c[20 * stride] = CLIP(t20 + t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  738|  1.93M|    c[21 * stride] = CLIP(t21 + t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  739|  1.93M|    c[22 * stride] = CLIP(t22 + t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  740|  1.93M|    c[23 * stride] = CLIP(t23 + t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  741|  1.93M|    c[24 * stride] = CLIP(t24 + t39);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  742|  1.93M|    c[25 * stride] = CLIP(t25 + t38a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  743|  1.93M|    c[26 * stride] = CLIP(t26 + t37);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  744|  1.93M|    c[27 * stride] = CLIP(t27 + t36a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  745|  1.93M|    c[28 * stride] = CLIP(t28 + t35);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  746|  1.93M|    c[29 * stride] = CLIP(t29 + t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  747|  1.93M|    c[30 * stride] = CLIP(t30 + t33);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  748|  1.93M|    c[31 * stride] = CLIP(t31 + t32a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  749|  1.93M|    c[32 * stride] = CLIP(t31 - t32a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  750|  1.93M|    c[33 * stride] = CLIP(t30 - t33);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  751|  1.93M|    c[34 * stride] = CLIP(t29 - t34a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  752|  1.93M|    c[35 * stride] = CLIP(t28 - t35);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  753|  1.93M|    c[36 * stride] = CLIP(t27 - t36a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  754|  1.93M|    c[37 * stride] = CLIP(t26 - t37);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  755|  1.93M|    c[38 * stride] = CLIP(t25 - t38a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  756|  1.93M|    c[39 * stride] = CLIP(t24 - t39);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  757|  1.93M|    c[40 * stride] = CLIP(t23 - t40a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  758|  1.93M|    c[41 * stride] = CLIP(t22 - t41);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  759|  1.93M|    c[42 * stride] = CLIP(t21 - t42a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  760|  1.93M|    c[43 * stride] = CLIP(t20 - t43);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  761|  1.93M|    c[44 * stride] = CLIP(t19 - t44a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  762|  1.93M|    c[45 * stride] = CLIP(t18 - t45);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  763|  1.93M|    c[46 * stride] = CLIP(t17 - t46a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  764|  1.93M|    c[47 * stride] = CLIP(t16 - t47);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  765|  1.93M|    c[48 * stride] = CLIP(t15 - t48);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  766|  1.93M|    c[49 * stride] = CLIP(t14 - t49a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  767|  1.93M|    c[50 * stride] = CLIP(t13 - t50);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  768|  1.93M|    c[51 * stride] = CLIP(t12 - t51a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  769|  1.93M|    c[52 * stride] = CLIP(t11 - t52);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  770|  1.93M|    c[53 * stride] = CLIP(t10 - t53a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  771|  1.93M|    c[54 * stride] = CLIP(t9  - t54);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  772|  1.93M|    c[55 * stride] = CLIP(t8  - t55a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  773|  1.93M|    c[56 * stride] = CLIP(t7  - t56);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  774|  1.93M|    c[57 * stride] = CLIP(t6  - t57a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  775|  1.93M|    c[58 * stride] = CLIP(t5  - t58);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  776|  1.93M|    c[59 * stride] = CLIP(t4  - t59a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  777|  1.93M|    c[60 * stride] = CLIP(t3  - t60);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  778|  1.93M|    c[61 * stride] = CLIP(t2  - t61a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  779|  1.93M|    c[62 * stride] = CLIP(t1  - t62);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  780|  1.93M|    c[63 * stride] = CLIP(t0  - t63a);
  ------------------
  |  |   37|  1.93M|#define CLIP(a) iclip(a, min, max)
  ------------------
  781|  1.93M|}

dav1d_itx_dsp_init_8bpc:
  220|  3.66k|COLD void bitfn(dav1d_itx_dsp_init)(Dav1dInvTxfmDSPContext *const c, int bpc) {
  221|  3.66k|#define assign_itx_all_fn64(w, h, pfx) \
  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  224|       |
  225|  3.66k|#define assign_itx_all_fn32(w, h, pfx) \
  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  229|       |
  230|  3.66k|#define assign_itx_all_fn16(w, h, pfx) \
  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  252|       |
  253|  3.66k|#define assign_itx_all_fn84(w, h, pfx) \
  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  263|  3.66k|
  264|  3.66k|#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
  265|  3.66k|  ARCH_AARCH64 || \
  266|  3.66k|  (ARCH_ARM && (defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32))) \
  267|  3.66k|))
  268|  3.66k|    c->itxfm_add[TX_4X4][WHT_WHT] = inv_txfm_add_wht_wht_4x4_c;
  269|  3.66k|#endif
  270|  3.66k|    assign_itx_all_fn84( 4,  4, );
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  271|  3.66k|    assign_itx_all_fn84( 4,  8, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  272|  3.66k|    assign_itx_all_fn84( 4, 16, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  273|  3.66k|    assign_itx_all_fn84( 8,  4, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  274|  3.66k|    assign_itx_all_fn84( 8,  8, );
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  275|  3.66k|    assign_itx_all_fn84( 8, 16, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  276|  3.66k|    assign_itx_all_fn32( 8, 32, R);
  ------------------
  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  277|  3.66k|    assign_itx_all_fn84(16,  4, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  278|  3.66k|    assign_itx_all_fn84(16,  8, R);
  ------------------
  |  |  254|  3.66k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.66k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.66k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.66k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.66k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  279|  3.66k|    assign_itx_all_fn16(16, 16, );
  ------------------
  |  |  231|  3.66k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  ------------------
  |  |  232|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  233|  3.66k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  234|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  235|  3.66k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  236|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  237|  3.66k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  238|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  239|  3.66k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  240|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  241|  3.66k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  242|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  243|  3.66k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  244|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  245|  3.66k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  246|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  247|  3.66k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  248|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  249|  3.66k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  250|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  251|  3.66k|        inv_txfm_add_identity_dct_##w##x##h##_c
  ------------------
  280|  3.66k|    assign_itx_all_fn32(16, 32, R);
  ------------------
  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  281|  3.66k|    assign_itx_all_fn64(16, 64, R);
  ------------------
  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  282|  3.66k|    assign_itx_all_fn32(32,  8, R);
  ------------------
  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  283|  3.66k|    assign_itx_all_fn32(32, 16, R);
  ------------------
  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  284|  3.66k|    assign_itx_all_fn32(32, 32, );
  ------------------
  |  |  226|  3.66k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.66k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  285|  3.66k|    assign_itx_all_fn64(32, 64, R);
  ------------------
  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  286|  3.66k|    assign_itx_all_fn64(64, 16, R);
  ------------------
  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  287|  3.66k|    assign_itx_all_fn64(64, 32, R);
  ------------------
  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  288|  3.66k|    assign_itx_all_fn64(64, 64, );
  ------------------
  |  |  222|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.66k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  289|       |
  290|  3.66k|    int all_simd = 0;
  291|  3.66k|#if HAVE_ASM
  292|       |#if ARCH_AARCH64 || ARCH_ARM
  293|       |    itx_dsp_init_arm(c, bpc, &all_simd);
  294|       |#endif
  295|       |#if ARCH_LOONGARCH64
  296|       |    itx_dsp_init_loongarch(c, bpc);
  297|       |#endif
  298|       |#if ARCH_PPC64LE
  299|       |    itx_dsp_init_ppc(c, bpc);
  300|       |#endif
  301|       |#if ARCH_RISCV
  302|       |    itx_dsp_init_riscv(c, bpc);
  303|       |#endif
  304|  3.66k|#if ARCH_X86
  305|  3.66k|    itx_dsp_init_x86(c, bpc, &all_simd);
  306|  3.66k|#endif
  307|  3.66k|#endif
  308|       |
  309|  3.66k|    if (!all_simd)
  ------------------
  |  Branch (309:9): [True: 0, False: 3.66k]
  ------------------
  310|      0|        dav1d_init_last_nonzero_col_from_eob_tables();
  311|  3.66k|}
itx_tmpl.c:inv_txfm_add_c:
   47|   174k|{
   48|   174k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
   49|   174k|    const int w = 4 * t_dim->w, h = 4 * t_dim->h;
   50|   174k|    const int has_dconly = txtp == DCT_DCT;
   51|   174k|    assert(w >= 4 && w <= 64);
  ------------------
  |  Branch (51:5): [True: 174k, False: 0]
  |  Branch (51:5): [True: 174k, False: 0]
  ------------------
   52|   174k|    assert(h >= 4 && h <= 64);
  ------------------
  |  Branch (52:5): [True: 174k, False: 0]
  |  Branch (52:5): [True: 174k, False: 0]
  ------------------
   53|   174k|    assert(eob >= 0);
  ------------------
  |  Branch (53:5): [True: 174k, False: 0]
  ------------------
   54|       |
   55|   174k|    const int is_rect2 = w * 2 == h || h * 2 == w;
  ------------------
  |  Branch (55:26): [True: 21.4k, False: 153k]
  |  Branch (55:40): [True: 56.2k, False: 96.8k]
  ------------------
   56|   174k|    const int rnd = (1 << shift) >> 1;
   57|       |
   58|   174k|    if (eob < has_dconly) {
  ------------------
  |  Branch (58:9): [True: 51.2k, False: 123k]
  ------------------
   59|  51.2k|        int dc = coeff[0];
   60|  51.2k|        coeff[0] = 0;
   61|  51.2k|        if (is_rect2)
  ------------------
  |  Branch (61:13): [True: 20.1k, False: 31.1k]
  ------------------
   62|  20.1k|            dc = (dc * 181 + 128) >> 8;
   63|  51.2k|        dc = (dc * 181 + 128) >> 8;
   64|  51.2k|        dc = (dc + rnd) >> shift;
   65|  51.2k|        dc = (dc * 181 + 128 + 2048) >> 12;
   66|  1.88M|        for (int y = 0; y < h; y++, dst += PXSTRIDE(stride))
  ------------------
  |  |   53|  1.83M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (66:25): [True: 1.83M, False: 51.2k]
  ------------------
   67|  83.8M|            for (int x = 0; x < w; x++)
  ------------------
  |  Branch (67:29): [True: 82.0M, False: 1.83M]
  ------------------
   68|  82.0M|                dst[x] = iclip_pixel(dst[x] + dc);
  ------------------
  |  |   49|  82.0M|#define iclip_pixel iclip_u8
  ------------------
   69|  51.2k|        return;
   70|  51.2k|    }
   71|       |
   72|   123k|    const uint8_t *const txtps = dav1d_tx1d_types[txtp];
   73|   123k|    const itx_1d_fn first_1d_fn = dav1d_tx1d_fns[t_dim->lw][txtps[0]];
   74|   123k|    const itx_1d_fn second_1d_fn = dav1d_tx1d_fns[t_dim->lh][txtps[1]];
   75|   123k|    const int sh = imin(h, 32), sw = imin(w, 32);
   76|   123k|#if BITDEPTH == 8
   77|   123k|    const int row_clip_min = INT16_MIN;
   78|   123k|    const int col_clip_min = INT16_MIN;
   79|       |#else
   80|       |    const int row_clip_min = (int) ((unsigned) ~bitdepth_max << 7);
   81|       |    const int col_clip_min = (int) ((unsigned) ~bitdepth_max << 5);
   82|       |#endif
   83|   123k|    const int row_clip_max = ~row_clip_min;
   84|   123k|    const int col_clip_max = ~col_clip_min;
   85|       |
   86|   123k|    int32_t tmp[64 * 64], *c = tmp;
   87|   123k|    int last_nonzero_col; // in first 1d itx
   88|   123k|    if (txtps[1] == IDENTITY && txtps[0] != IDENTITY) {
  ------------------
  |  Branch (88:9): [True: 0, False: 123k]
  |  Branch (88:33): [True: 0, False: 0]
  ------------------
   89|      0|        last_nonzero_col = imin(sh - 1, eob);
   90|   123k|    } else if (txtps[0] == IDENTITY && txtps[1] != IDENTITY) {
  ------------------
  |  Branch (90:16): [True: 0, False: 123k]
  |  Branch (90:40): [True: 0, False: 0]
  ------------------
   91|      0|        last_nonzero_col = eob >> (t_dim->lw + 2);
   92|   123k|    } else {
   93|   123k|        last_nonzero_col = dav1d_last_nonzero_col_from_eob[tx][eob];
   94|   123k|    }
   95|   123k|    assert(last_nonzero_col < sh);
  ------------------
  |  Branch (95:5): [True: 123k, False: 0]
  ------------------
   96|  1.19M|    for (int y = 0; y <= last_nonzero_col; y++, c += w) {
  ------------------
  |  Branch (96:21): [True: 1.06M, False: 123k]
  ------------------
   97|  1.06M|        if (is_rect2)
  ------------------
  |  Branch (97:13): [True: 413k, False: 655k]
  ------------------
   98|  12.2M|            for (int x = 0; x < sw; x++)
  ------------------
  |  Branch (98:29): [True: 11.8M, False: 413k]
  ------------------
   99|  11.8M|                c[x] = (coeff[y + x * sh] * 181 + 128) >> 8;
  100|   655k|        else
  101|  21.3M|            for (int x = 0; x < sw; x++)
  ------------------
  |  Branch (101:29): [True: 20.7M, False: 655k]
  ------------------
  102|  20.7M|                c[x] = coeff[y + x * sh];
  103|  1.06M|        first_1d_fn(c, 1, row_clip_min, row_clip_max);
  104|  1.06M|    }
  105|   123k|    if (last_nonzero_col + 1 < sh)
  ------------------
  |  Branch (105:9): [True: 113k, False: 10.1k]
  ------------------
  106|   113k|        memset(c, 0, sizeof(*c) * (sh - last_nonzero_col - 1) * w);
  107|       |
  108|   123k|    memset(coeff, 0, sizeof(*coeff) * sw * sh);
  109|   139M|    for (int i = 0; i < w * sh; i++)
  ------------------
  |  Branch (109:21): [True: 139M, False: 123k]
  ------------------
  110|   139M|        tmp[i] = iclip((tmp[i] + rnd) >> shift, col_clip_min, col_clip_max);
  111|       |
  112|  5.07M|    for (int x = 0; x < w; x++)
  ------------------
  |  Branch (112:21): [True: 4.95M, False: 123k]
  ------------------
  113|  4.95M|        second_1d_fn(&tmp[x], w, col_clip_min, col_clip_max);
  114|       |
  115|   123k|    c = tmp;
  116|  4.40M|    for (int y = 0; y < h; y++, dst += PXSTRIDE(stride))
  ------------------
  |  |   53|  4.28M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (116:21): [True: 4.28M, False: 123k]
  ------------------
  117|   193M|        for (int x = 0; x < w; x++)
  ------------------
  |  Branch (117:25): [True: 189M, False: 4.28M]
  ------------------
  118|   189M|            dst[x] = iclip_pixel(dst[x] + ((*c++ + 8) >> 4));
  ------------------
  |  |   49|   189M|#define iclip_pixel iclip_u8
  ------------------
  119|   123k|}
itx_tmpl.c:inv_txfm_add_dct_dct_16x32_c:
  127|  18.3k|                                               HIGHBD_DECL_SUFFIX) \
  128|  18.3k|{ \
  129|  18.3k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  18.3k|                   HIGHBD_TAIL_SUFFIX); \
  131|  18.3k|}
itx_tmpl.c:inv_txfm_add_dct_dct_16x64_c:
  127|  2.70k|                                               HIGHBD_DECL_SUFFIX) \
  128|  2.70k|{ \
  129|  2.70k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  2.70k|                   HIGHBD_TAIL_SUFFIX); \
  131|  2.70k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x16_c:
  127|  36.9k|                                               HIGHBD_DECL_SUFFIX) \
  128|  36.9k|{ \
  129|  36.9k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  36.9k|                   HIGHBD_TAIL_SUFFIX); \
  131|  36.9k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x32_c:
  127|  57.5k|                                               HIGHBD_DECL_SUFFIX) \
  128|  57.5k|{ \
  129|  57.5k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  57.5k|                   HIGHBD_TAIL_SUFFIX); \
  131|  57.5k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x64_c:
  127|  3.06k|                                               HIGHBD_DECL_SUFFIX) \
  128|  3.06k|{ \
  129|  3.06k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  3.06k|                   HIGHBD_TAIL_SUFFIX); \
  131|  3.06k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x16_c:
  127|  4.75k|                                               HIGHBD_DECL_SUFFIX) \
  128|  4.75k|{ \
  129|  4.75k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  4.75k|                   HIGHBD_TAIL_SUFFIX); \
  131|  4.75k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x32_c:
  127|  19.2k|                                               HIGHBD_DECL_SUFFIX) \
  128|  19.2k|{ \
  129|  19.2k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  19.2k|                   HIGHBD_TAIL_SUFFIX); \
  131|  19.2k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x64_c:
  127|  31.8k|                                               HIGHBD_DECL_SUFFIX) \
  128|  31.8k|{ \
  129|  31.8k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  31.8k|                   HIGHBD_TAIL_SUFFIX); \
  131|  31.8k|}
dav1d_itx_dsp_init_16bpc:
  220|  4.99k|COLD void bitfn(dav1d_itx_dsp_init)(Dav1dInvTxfmDSPContext *const c, int bpc) {
  221|  4.99k|#define assign_itx_all_fn64(w, h, pfx) \
  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  224|       |
  225|  4.99k|#define assign_itx_all_fn32(w, h, pfx) \
  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  229|       |
  230|  4.99k|#define assign_itx_all_fn16(w, h, pfx) \
  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  252|       |
  253|  4.99k|#define assign_itx_all_fn84(w, h, pfx) \
  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  263|  4.99k|
  264|  4.99k|#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
  265|  4.99k|  ARCH_AARCH64 || \
  266|  4.99k|  (ARCH_ARM && (defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32))) \
  267|  4.99k|))
  268|  4.99k|    c->itxfm_add[TX_4X4][WHT_WHT] = inv_txfm_add_wht_wht_4x4_c;
  269|  4.99k|#endif
  270|  4.99k|    assign_itx_all_fn84( 4,  4, );
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  271|  4.99k|    assign_itx_all_fn84( 4,  8, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  272|  4.99k|    assign_itx_all_fn84( 4, 16, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  273|  4.99k|    assign_itx_all_fn84( 8,  4, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  274|  4.99k|    assign_itx_all_fn84( 8,  8, );
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  275|  4.99k|    assign_itx_all_fn84( 8, 16, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  276|  4.99k|    assign_itx_all_fn32( 8, 32, R);
  ------------------
  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  277|  4.99k|    assign_itx_all_fn84(16,  4, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  278|  4.99k|    assign_itx_all_fn84(16,  8, R);
  ------------------
  |  |  254|  4.99k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  4.99k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  4.99k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  4.99k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  4.99k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  279|  4.99k|    assign_itx_all_fn16(16, 16, );
  ------------------
  |  |  231|  4.99k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  ------------------
  |  |  232|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  233|  4.99k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  234|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  235|  4.99k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  236|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  237|  4.99k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  238|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  239|  4.99k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  240|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  241|  4.99k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  242|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  243|  4.99k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  244|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  245|  4.99k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  246|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  247|  4.99k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  248|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  249|  4.99k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  250|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  251|  4.99k|        inv_txfm_add_identity_dct_##w##x##h##_c
  ------------------
  280|  4.99k|    assign_itx_all_fn32(16, 32, R);
  ------------------
  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  281|  4.99k|    assign_itx_all_fn64(16, 64, R);
  ------------------
  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  282|  4.99k|    assign_itx_all_fn32(32,  8, R);
  ------------------
  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  283|  4.99k|    assign_itx_all_fn32(32, 16, R);
  ------------------
  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  284|  4.99k|    assign_itx_all_fn32(32, 32, );
  ------------------
  |  |  226|  4.99k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  4.99k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  285|  4.99k|    assign_itx_all_fn64(32, 64, R);
  ------------------
  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  286|  4.99k|    assign_itx_all_fn64(64, 16, R);
  ------------------
  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  287|  4.99k|    assign_itx_all_fn64(64, 32, R);
  ------------------
  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  288|  4.99k|    assign_itx_all_fn64(64, 64, );
  ------------------
  |  |  222|  4.99k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  4.99k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  289|       |
  290|  4.99k|    int all_simd = 0;
  291|  4.99k|#if HAVE_ASM
  292|       |#if ARCH_AARCH64 || ARCH_ARM
  293|       |    itx_dsp_init_arm(c, bpc, &all_simd);
  294|       |#endif
  295|       |#if ARCH_LOONGARCH64
  296|       |    itx_dsp_init_loongarch(c, bpc);
  297|       |#endif
  298|       |#if ARCH_PPC64LE
  299|       |    itx_dsp_init_ppc(c, bpc);
  300|       |#endif
  301|       |#if ARCH_RISCV
  302|       |    itx_dsp_init_riscv(c, bpc);
  303|       |#endif
  304|  4.99k|#if ARCH_X86
  305|  4.99k|    itx_dsp_init_x86(c, bpc, &all_simd);
  306|  4.99k|#endif
  307|  4.99k|#endif
  308|       |
  309|  4.99k|    if (!all_simd)
  ------------------
  |  Branch (309:9): [True: 2.52k, False: 2.47k]
  ------------------
  310|  2.52k|        dav1d_init_last_nonzero_col_from_eob_tables();
  311|  4.99k|}

dav1d_copy_lpf_8bpc:
  106|  46.3k|{
  107|  46.3k|    const int have_tt = f->c->n_tc > 1;
  108|  46.3k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  109|  46.3k|    const int offset = 8 * !!sby;
  110|  46.3k|    const ptrdiff_t *const src_stride = f->cur.stride;
  111|  46.3k|    const ptrdiff_t *const lr_stride = f->sr_cur.p.stride;
  112|  46.3k|    const int tt_off = have_tt * sby * (4 << f->seq_hdr->sb128);
  113|  46.3k|    pixel *const dst[3] = {
  114|  46.3k|        f->lf.lr_lpf_line[0] + tt_off * PXSTRIDE(lr_stride[0]),
  ------------------
  |  |   53|  46.3k|#define PXSTRIDE(x) (x)
  ------------------
  115|  46.3k|        f->lf.lr_lpf_line[1] + tt_off * PXSTRIDE(lr_stride[1]),
  ------------------
  |  |   53|  46.3k|#define PXSTRIDE(x) (x)
  ------------------
  116|  46.3k|        f->lf.lr_lpf_line[2] + tt_off * PXSTRIDE(lr_stride[1])
  ------------------
  |  |   53|  46.3k|#define PXSTRIDE(x) (x)
  ------------------
  117|  46.3k|    };
  118|       |
  119|       |    // TODO Also check block level restore type to reduce copying.
  120|  46.3k|    const int restore_planes = f->lf.restore_planes;
  121|       |
  122|  46.3k|    if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (122:9): [True: 34.6k, False: 11.7k]
  |  Branch (122:29): [True: 8.18k, False: 3.53k]
  ------------------
  123|  42.8k|        const int h = f->cur.p.h;
  124|  42.8k|        const int w = f->bw << 2;
  125|  42.8k|        const int row_h = imin((sby + 1) << (6 + f->seq_hdr->sb128), h - 1);
  126|  42.8k|        const int y_stripe = (sby << (6 + f->seq_hdr->sb128)) - offset;
  127|  42.8k|        if (restore_planes & LR_RESTORE_Y || !resize)
  ------------------
  |  Branch (127:13): [True: 16.7k, False: 26.1k]
  |  Branch (127:46): [True: 24.4k, False: 1.70k]
  ------------------
  128|  41.1k|            backup_lpf(f, dst[0], lr_stride[0],
  129|  41.1k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  ------------------
  |  |   53|  41.1k|#define PXSTRIDE(x) (x)
  ------------------
  130|  41.1k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 1);
  131|  42.8k|        if (have_tt && resize) {
  ------------------
  |  Branch (131:13): [True: 0, False: 42.8k]
  |  Branch (131:24): [True: 0, False: 0]
  ------------------
  132|      0|            const ptrdiff_t cdef_off_y = sby * 4 * PXSTRIDE(src_stride[0]);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
  133|      0|            backup_lpf(f, f->lf.cdef_lpf_line[0] + cdef_off_y, src_stride[0],
  134|      0|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
  135|      0|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 0);
  136|      0|        }
  137|  42.8k|    }
  138|  46.3k|    if ((f->seq_hdr->cdef || restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) &&
  ------------------
  |  Branch (138:10): [True: 34.6k, False: 11.7k]
  |  Branch (138:30): [True: 4.33k, False: 7.37k]
  ------------------
  139|  38.9k|        f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400)
  ------------------
  |  Branch (139:9): [True: 18.5k, False: 20.4k]
  ------------------
  140|  18.5k|    {
  141|  18.5k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  142|  18.5k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  143|  18.5k|        const int h = (f->cur.p.h + ss_ver) >> ss_ver;
  144|  18.5k|        const int w = f->bw << (2 - ss_hor);
  145|  18.5k|        const int row_h = imin((sby + 1) << ((6 - ss_ver) + f->seq_hdr->sb128), h - 1);
  146|  18.5k|        const int offset_uv = offset >> ss_ver;
  147|  18.5k|        const int y_stripe = (sby << ((6 - ss_ver) + f->seq_hdr->sb128)) - offset_uv;
  148|  18.5k|        const ptrdiff_t cdef_off_uv = sby * 4 * PXSTRIDE(src_stride[1]);
  ------------------
  |  |   53|  18.5k|#define PXSTRIDE(x) (x)
  ------------------
  149|  18.5k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_U) {
  ------------------
  |  Branch (149:13): [True: 14.2k, False: 4.33k]
  |  Branch (149:33): [True: 846, False: 3.49k]
  ------------------
  150|  15.0k|            if (restore_planes & LR_RESTORE_U || !resize)
  ------------------
  |  Branch (150:17): [True: 2.78k, False: 12.2k]
  |  Branch (150:50): [True: 11.5k, False: 759]
  ------------------
  151|  14.3k|                backup_lpf(f, dst[1], lr_stride[1],
  152|  14.3k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  14.3k|#define PXSTRIDE(x) (x)
  ------------------
  153|  14.3k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  154|  14.3k|                           row_h, w, h, ss_hor, 1);
  155|  15.0k|            if (have_tt && resize)
  ------------------
  |  Branch (155:17): [True: 0, False: 15.0k]
  |  Branch (155:28): [True: 0, False: 0]
  ------------------
  156|      0|                backup_lpf(f, f->lf.cdef_lpf_line[1] + cdef_off_uv, src_stride[1],
  157|      0|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
  158|      0|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  159|      0|                           row_h, w, h, ss_hor, 0);
  160|  15.0k|        }
  161|  18.5k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_V) {
  ------------------
  |  Branch (161:13): [True: 14.2k, False: 4.33k]
  |  Branch (161:33): [True: 3.92k, False: 412]
  ------------------
  162|  18.1k|            if (restore_planes & LR_RESTORE_V || !resize)
  ------------------
  |  Branch (162:17): [True: 5.86k, False: 12.2k]
  |  Branch (162:50): [True: 11.0k, False: 1.28k]
  ------------------
  163|  16.8k|                backup_lpf(f, dst[2], lr_stride[1],
  164|  16.8k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  16.8k|#define PXSTRIDE(x) (x)
  ------------------
  165|  16.8k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  166|  16.8k|                           row_h, w, h, ss_hor, 1);
  167|  18.1k|            if (have_tt && resize)
  ------------------
  |  Branch (167:17): [True: 0, False: 18.1k]
  |  Branch (167:28): [True: 0, False: 0]
  ------------------
  168|      0|                backup_lpf(f, f->lf.cdef_lpf_line[2] + cdef_off_uv, src_stride[1],
  169|      0|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
  170|      0|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  171|      0|                           row_h, w, h, ss_hor, 0);
  172|  18.1k|        }
  173|  18.5k|    }
  174|  46.3k|}
dav1d_loopfilter_sbrow_cols_8bpc:
  320|  34.2k|{
  321|  34.2k|    int x, have_left;
  322|       |    // Don't filter outside the frame
  323|  34.2k|    const int is_sb64 = !f->seq_hdr->sb128;
  324|  34.2k|    const int starty4 = (sby & is_sb64) << 4;
  325|  34.2k|    const int sbsz = 32 >> is_sb64;
  326|  34.2k|    const int sbl2 = 5 - is_sb64;
  327|  34.2k|    const int halign = (f->bh + 31) & ~31;
  328|  34.2k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  329|  34.2k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  330|  34.2k|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  331|  34.2k|    const unsigned vmax = 1U << vmask, hmax = 1U << hmask;
  332|  34.2k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  333|  34.2k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  334|       |
  335|       |    // fix lpf strength at tile col boundaries
  336|  34.2k|    const uint8_t *lpf_y = &f->lf.tx_lpf_right_edge[0][sby << sbl2];
  337|  34.2k|    const uint8_t *lpf_uv = &f->lf.tx_lpf_right_edge[1][sby << (sbl2 - ss_ver)];
  338|  35.3k|    for (int tile_col = 1;; tile_col++) {
  339|  35.3k|        x = f->frame_hdr->tiling.col_start_sb[tile_col];
  340|  35.3k|        if ((x << sbl2) >= f->bw) break;
  ------------------
  |  Branch (340:13): [True: 34.2k, False: 1.06k]
  ------------------
  341|  1.06k|        const int bx4 = x & is_sb64 ? 16 : 0, cbx4 = bx4 >> ss_hor;
  ------------------
  |  Branch (341:25): [True: 413, False: 648]
  ------------------
  342|  1.06k|        x >>= is_sb64;
  343|       |
  344|  1.06k|        uint16_t (*const y_hmask)[2] = lflvl[x].filter_y[0][bx4];
  345|  27.2k|        for (unsigned y = starty4, mask = 1 << y; y < endy4; y++, mask <<= 1) {
  ------------------
  |  Branch (345:51): [True: 26.2k, False: 1.06k]
  ------------------
  346|  26.2k|            const int sidx = mask >= 0x10000U;
  347|  26.2k|            const unsigned smask = mask >> (sidx << 4);
  348|  26.2k|            const int idx = 2 * !!(y_hmask[2][sidx] & smask) +
  349|  26.2k|                                !!(y_hmask[1][sidx] & smask);
  350|  26.2k|            y_hmask[2][sidx] &= ~smask;
  351|  26.2k|            y_hmask[1][sidx] &= ~smask;
  352|  26.2k|            y_hmask[0][sidx] &= ~smask;
  353|  26.2k|            y_hmask[imin(idx, lpf_y[y - starty4])][sidx] |= smask;
  354|  26.2k|        }
  355|       |
  356|  1.06k|        if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (356:13): [True: 378, False: 683]
  ------------------
  357|    378|            uint16_t (*const uv_hmask)[2] = lflvl[x].filter_uv[0][cbx4];
  358|  4.63k|            for (unsigned y = starty4 >> ss_ver, uv_mask = 1 << y; y < uv_endy4;
  ------------------
  |  Branch (358:68): [True: 4.26k, False: 378]
  ------------------
  359|  4.26k|                 y++, uv_mask <<= 1)
  360|  4.26k|            {
  361|  4.26k|                const int sidx = uv_mask >= vmax;
  362|  4.26k|                const unsigned smask = uv_mask >> (sidx << (4 - ss_ver));
  363|  4.26k|                const int idx = !!(uv_hmask[1][sidx] & smask);
  364|  4.26k|                uv_hmask[1][sidx] &= ~smask;
  365|  4.26k|                uv_hmask[0][sidx] &= ~smask;
  366|  4.26k|                uv_hmask[imin(idx, lpf_uv[y - (starty4 >> ss_ver)])][sidx] |= smask;
  367|  4.26k|            }
  368|    378|        }
  369|  1.06k|        lpf_y  += halign;
  370|  1.06k|        lpf_uv += halign >> ss_ver;
  371|  1.06k|    }
  372|       |
  373|       |    // fix lpf strength at tile row boundaries
  374|  34.2k|    if (start_of_tile_row) {
  ------------------
  |  Branch (374:9): [True: 390, False: 33.8k]
  ------------------
  375|    390|        const BlockContext *a;
  376|    390|        for (x = 0, a = &f->a[f->sb128w * (start_of_tile_row - 1)];
  377|  1.99k|             x < f->sb128w; x++, a++)
  ------------------
  |  Branch (377:14): [True: 1.60k, False: 390]
  ------------------
  378|  1.60k|        {
  379|  1.60k|            uint16_t (*const y_vmask)[2] = lflvl[x].filter_y[1][starty4];
  380|  1.60k|            const unsigned w = imin(32, f->w4 - (x << 5));
  381|  45.7k|            for (unsigned mask = 1, i = 0; i < w; mask <<= 1, i++) {
  ------------------
  |  Branch (381:44): [True: 44.1k, False: 1.60k]
  ------------------
  382|  44.1k|                const int sidx = mask >= 0x10000U;
  383|  44.1k|                const unsigned smask = mask >> (sidx << 4);
  384|  44.1k|                const int idx = 2 * !!(y_vmask[2][sidx] & smask) +
  385|  44.1k|                                    !!(y_vmask[1][sidx] & smask);
  386|  44.1k|                y_vmask[2][sidx] &= ~smask;
  387|  44.1k|                y_vmask[1][sidx] &= ~smask;
  388|  44.1k|                y_vmask[0][sidx] &= ~smask;
  389|  44.1k|                y_vmask[imin(idx, a->tx_lpf_y[i])][sidx] |= smask;
  390|  44.1k|            }
  391|       |
  392|  1.60k|            if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (392:17): [True: 1.22k, False: 376]
  ------------------
  393|  1.22k|                const unsigned cw = (w + ss_hor) >> ss_hor;
  394|  1.22k|                uint16_t (*const uv_vmask)[2] = lflvl[x].filter_uv[1][starty4 >> ss_ver];
  395|  32.6k|                for (unsigned uv_mask = 1, i = 0; i < cw; uv_mask <<= 1, i++) {
  ------------------
  |  Branch (395:51): [True: 31.4k, False: 1.22k]
  ------------------
  396|  31.4k|                    const int sidx = uv_mask >= hmax;
  397|  31.4k|                    const unsigned smask = uv_mask >> (sidx << (4 - ss_hor));
  398|  31.4k|                    const int idx = !!(uv_vmask[1][sidx] & smask);
  399|  31.4k|                    uv_vmask[1][sidx] &= ~smask;
  400|  31.4k|                    uv_vmask[0][sidx] &= ~smask;
  401|  31.4k|                    uv_vmask[imin(idx, a->tx_lpf_uv[i])][sidx] |= smask;
  402|  31.4k|                }
  403|  1.22k|            }
  404|  1.60k|        }
  405|    390|    }
  406|       |
  407|  34.2k|    pixel *ptr;
  408|  34.2k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  409|  81.7k|    for (ptr = p[0], have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (409:44): [True: 47.4k, False: 34.2k]
  ------------------
  410|  47.4k|         x++, have_left = 1, ptr += 128, level_ptr += 32)
  411|  47.4k|    {
  412|  47.4k|        filter_plane_cols_y(f, have_left, level_ptr, f->b4_stride,
  413|  47.4k|                            lflvl[x].filter_y[0], ptr, f->cur.stride[0],
  414|  47.4k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  415|  47.4k|    }
  416|       |
  417|  34.2k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (417:9): [True: 24.4k, False: 9.80k]
  |  Branch (417:46): [True: 19.6k, False: 4.84k]
  ------------------
  418|  19.6k|        return;
  419|       |
  420|  14.6k|    ptrdiff_t uv_off;
  421|  14.6k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  422|  36.8k|    for (uv_off = 0, have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (422:44): [True: 22.1k, False: 14.6k]
  ------------------
  423|  22.1k|         x++, have_left = 1, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  424|  22.1k|    {
  425|  22.1k|        filter_plane_cols_uv(f, have_left, level_ptr, f->b4_stride,
  426|  22.1k|                             lflvl[x].filter_uv[0],
  427|  22.1k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  428|  22.1k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  429|  22.1k|                             starty4 >> ss_ver, uv_endy4, ss_ver);
  430|  22.1k|    }
  431|  14.6k|}
dav1d_loopfilter_sbrow_rows_8bpc:
  436|  34.2k|{
  437|  34.2k|    int x;
  438|       |    // Don't filter outside the frame
  439|  34.2k|    const int have_top = sby > 0;
  440|  34.2k|    const int is_sb64 = !f->seq_hdr->sb128;
  441|  34.2k|    const int starty4 = (sby & is_sb64) << 4;
  442|  34.2k|    const int sbsz = 32 >> is_sb64;
  443|  34.2k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  444|  34.2k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  445|  34.2k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  446|  34.2k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  447|       |
  448|  34.2k|    pixel *ptr;
  449|  34.2k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  450|  81.7k|    for (ptr = p[0], x = 0; x < f->sb128w; x++, ptr += 128, level_ptr += 32) {
  ------------------
  |  Branch (450:29): [True: 47.4k, False: 34.2k]
  ------------------
  451|  47.4k|        filter_plane_rows_y(f, have_top, level_ptr, f->b4_stride,
  452|  47.4k|                            lflvl[x].filter_y[1], ptr, f->cur.stride[0],
  453|  47.4k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  454|  47.4k|    }
  455|       |
  456|  34.2k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (456:9): [True: 24.4k, False: 9.80k]
  |  Branch (456:46): [True: 19.6k, False: 4.84k]
  ------------------
  457|  19.6k|        return;
  458|       |
  459|  14.6k|    ptrdiff_t uv_off;
  460|  14.6k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  461|  36.8k|    for (uv_off = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (461:29): [True: 22.1k, False: 14.6k]
  ------------------
  462|  22.1k|         x++, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  463|  22.1k|    {
  464|  22.1k|        filter_plane_rows_uv(f, have_top, level_ptr, f->b4_stride,
  465|  22.1k|                             lflvl[x].filter_uv[1],
  466|  22.1k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  467|  22.1k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  468|  22.1k|                             starty4 >> ss_ver, uv_endy4, ss_hor);
  469|  22.1k|    }
  470|  14.6k|}
lf_apply_tmpl.c:backup_lpf:
   47|   129k|{
   48|   129k|    const int cdef_backup = !lr_backup;
   49|   129k|    const int dst_w = f->frame_hdr->super_res.enabled ?
  ------------------
  |  Branch (49:23): [True: 13.4k, False: 116k]
  ------------------
   50|   116k|                      (f->frame_hdr->width[1] + ss_hor) >> ss_hor : src_w;
   51|       |
   52|       |    // The first stripe of the frame is shorter by 8 luma pixel rows.
   53|   129k|    int stripe_h = ((64 << (cdef_backup & sb128)) - 8 * !row) >> ss_ver;
   54|   129k|    src += (stripe_h - 2) * PXSTRIDE(src_stride);
  ------------------
  |  |   53|   129k|#define PXSTRIDE(x) (x)
  ------------------
   55|       |
   56|   129k|    if (f->c->n_tc == 1) {
  ------------------
  |  Branch (56:9): [True: 129k, False: 0]
  ------------------
   57|   129k|        if (row) {
  ------------------
  |  Branch (57:13): [True: 107k, False: 22.7k]
  ------------------
   58|   107k|            const int top = 4 << sb128;
   59|       |            // Copy the top part of the stored loop filtered pixels from the
   60|       |            // previous sb row needed above the first stripe of this sb row.
   61|   107k|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  0],
  ------------------
  |  |   47|   107k|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  0],
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   62|   107k|                       &dst[PXSTRIDE(dst_stride) *  top],      dst_w);
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   63|   107k|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  1],
  ------------------
  |  |   47|   107k|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  1],
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   64|   107k|                       &dst[PXSTRIDE(dst_stride) * (top + 1)], dst_w);
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   65|   107k|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  2],
  ------------------
  |  |   47|   107k|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  2],
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   66|   107k|                       &dst[PXSTRIDE(dst_stride) * (top + 2)], dst_w);
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   67|   107k|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  3],
  ------------------
  |  |   47|   107k|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  3],
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   68|   107k|                       &dst[PXSTRIDE(dst_stride) * (top + 3)], dst_w);
  ------------------
  |  |   53|   107k|#define PXSTRIDE(x) (x)
  ------------------
   69|   107k|        }
   70|   129k|        dst += 4 * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|   129k|#define PXSTRIDE(x) (x)
  ------------------
   71|   129k|    }
   72|       |
   73|   129k|    if (lr_backup && (f->frame_hdr->width[0] != f->frame_hdr->width[1])) {
  ------------------
  |  Branch (73:9): [True: 129k, False: 0]
  |  Branch (73:22): [True: 10.6k, False: 119k]
  ------------------
   74|  24.0k|        while (row + stripe_h <= row_h) {
  ------------------
  |  Branch (74:16): [True: 13.3k, False: 10.6k]
  ------------------
   75|  13.3k|            const int n_lines = 4 - (row + stripe_h + 1 == h);
   76|  13.3k|            f->dsp->mc.resize(dst, dst_stride, src, src_stride,
   77|  13.3k|                              dst_w, n_lines, src_w, f->resize_step[ss_hor],
   78|  13.3k|                              f->resize_start[ss_hor] HIGHBD_CALL_SUFFIX);
   79|  13.3k|            row += stripe_h; // unmodified stripe_h for the 1st stripe
   80|  13.3k|            stripe_h = 64 >> ss_ver;
   81|  13.3k|            src += stripe_h * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  13.3k|#define PXSTRIDE(x) (x)
  ------------------
   82|  13.3k|            dst += n_lines * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  13.3k|#define PXSTRIDE(x) (x)
  ------------------
   83|  13.3k|            if (n_lines == 3) {
  ------------------
  |  Branch (83:17): [True: 2.05k, False: 11.3k]
  ------------------
   84|  2.05k|                pixel_copy(dst, &dst[-PXSTRIDE(dst_stride)], dst_w);
  ------------------
  |  |   47|  2.05k|#define pixel_copy memcpy
  ------------------
                              pixel_copy(dst, &dst[-PXSTRIDE(dst_stride)], dst_w);
  ------------------
  |  |   53|  2.05k|#define PXSTRIDE(x) (x)
  ------------------
   85|  2.05k|                dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  2.05k|#define PXSTRIDE(x) (x)
  ------------------
   86|  2.05k|            }
   87|  13.3k|        }
   88|   119k|    } else {
   89|   268k|        while (row + stripe_h <= row_h) {
  ------------------
  |  Branch (89:16): [True: 149k, False: 119k]
  ------------------
   90|   149k|            const int n_lines = 4 - (row + stripe_h + 1 == h);
   91|   745k|            for (int i = 0; i < 4; i++) {
  ------------------
  |  Branch (91:29): [True: 596k, False: 149k]
  ------------------
   92|   596k|                pixel_copy(dst, i == n_lines ? &dst[-PXSTRIDE(dst_stride)] :
  ------------------
  |  |   47|   596k|#define pixel_copy memcpy
  ------------------
                              pixel_copy(dst, i == n_lines ? &dst[-PXSTRIDE(dst_stride)] :
  ------------------
  |  |   53|  1.16k|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (92:33): [True: 1.16k, False: 595k]
  ------------------
   93|   596k|                                               src, src_w);
   94|   596k|                dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|   596k|#define PXSTRIDE(x) (x)
  ------------------
   95|   596k|                src += PXSTRIDE(src_stride);
  ------------------
  |  |   53|   596k|#define PXSTRIDE(x) (x)
  ------------------
   96|   596k|            }
   97|   149k|            row += stripe_h; // unmodified stripe_h for the 1st stripe
   98|   149k|            stripe_h = 64 >> ss_ver;
   99|   149k|            src += (stripe_h - 4) * PXSTRIDE(src_stride);
  ------------------
  |  |   53|   149k|#define PXSTRIDE(x) (x)
  ------------------
  100|   149k|        }
  101|   119k|    }
  102|   129k|}
lf_apply_tmpl.c:filter_plane_cols_y:
  184|  98.1k|{
  185|  98.1k|    const Dav1dDSPContext *const dsp = f->dsp;
  186|  98.1k|    const loopfilter_sb_fn loop_filter_sb = dsp->lf.loop_filter_sb[0][0];
  187|       |
  188|       |    // filter edges between columns (e.g. block1 | block2)
  189|  2.48M|    for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (189:21): [True: 2.39M, False: 98.1k]
  ------------------
  190|  2.39M|        if (!have_left && !x) continue;
  ------------------
  |  Branch (190:13): [True: 1.39M, False: 996k]
  |  Branch (190:27): [True: 66.0k, False: 1.32M]
  ------------------
  191|  2.32M|        uint32_t hmask[4];
  192|  2.32M|        if (!starty4) {
  ------------------
  |  Branch (192:13): [True: 2.01M, False: 308k]
  ------------------
  193|  2.01M|            hmask[0] = mask[x][0][0];
  194|  2.01M|            hmask[1] = mask[x][1][0];
  195|  2.01M|            hmask[2] = mask[x][2][0];
  196|  2.01M|            if (endy4 > 16) {
  ------------------
  |  Branch (196:17): [True: 1.58M, False: 433k]
  ------------------
  197|  1.58M|                hmask[0] |= (unsigned) mask[x][0][1] << 16;
  198|  1.58M|                hmask[1] |= (unsigned) mask[x][1][1] << 16;
  199|  1.58M|                hmask[2] |= (unsigned) mask[x][2][1] << 16;
  200|  1.58M|            }
  201|  2.01M|        } else {
  202|   308k|            hmask[0] = mask[x][0][1];
  203|   308k|            hmask[1] = mask[x][1][1];
  204|   308k|            hmask[2] = mask[x][2][1];
  205|   308k|        }
  206|  2.32M|        hmask[3] = 0;
  207|  2.32M|        loop_filter_sb(&dst[x * 4], ls, hmask,
  208|  2.32M|                       (const uint8_t(*)[4]) &lvl[x][0], b4_stride,
  209|  2.32M|                       &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  210|  2.32M|    }
  211|  98.1k|}
lf_apply_tmpl.c:filter_plane_cols_uv:
  253|  43.2k|{
  254|  43.2k|    const Dav1dDSPContext *const dsp = f->dsp;
  255|  43.2k|    const loopfilter_sb_fn loop_filter_sb = dsp->lf.loop_filter_sb[1][0];
  256|       |
  257|       |    // filter edges between columns (e.g. block1 | block2)
  258|   802k|    for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (258:21): [True: 758k, False: 43.2k]
  ------------------
  259|   758k|        if (!have_left && !x) continue;
  ------------------
  |  Branch (259:13): [True: 499k, False: 259k]
  |  Branch (259:27): [True: 31.6k, False: 467k]
  ------------------
  260|   727k|        uint32_t hmask[3];
  261|   727k|        if (!starty4) {
  ------------------
  |  Branch (261:13): [True: 647k, False: 79.9k]
  ------------------
  262|   647k|            hmask[0] = mask[x][0][0];
  263|   647k|            hmask[1] = mask[x][1][0];
  264|   647k|            if (endy4 > (16 >> ss_ver)) {
  ------------------
  |  Branch (264:17): [True: 518k, False: 128k]
  ------------------
  265|   518k|                hmask[0] |= (unsigned) mask[x][0][1] << (16 >> ss_ver);
  266|   518k|                hmask[1] |= (unsigned) mask[x][1][1] << (16 >> ss_ver);
  267|   518k|            }
  268|   647k|        } else {
  269|  79.9k|            hmask[0] = mask[x][0][1];
  270|  79.9k|            hmask[1] = mask[x][1][1];
  271|  79.9k|        }
  272|   727k|        hmask[2] = 0;
  273|   727k|        loop_filter_sb(&u[x * 4], ls, hmask,
  274|   727k|                       (const uint8_t(*)[4]) &lvl[x][2], b4_stride,
  275|   727k|                       &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  276|   727k|        loop_filter_sb(&v[x * 4], ls, hmask,
  277|   727k|                       (const uint8_t(*)[4]) &lvl[x][3], b4_stride,
  278|   727k|                       &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  279|   727k|    }
  280|  43.2k|}
lf_apply_tmpl.c:filter_plane_rows_y:
  221|  98.1k|{
  222|  98.1k|    const Dav1dDSPContext *const dsp = f->dsp;
  223|  98.1k|    const loopfilter_sb_fn loop_filter_sb = dsp->lf.loop_filter_sb[0][1];
  224|       |
  225|       |    //                                 block1
  226|       |    // filter edges between rows (e.g. ------)
  227|       |    //                                 block2
  228|  2.36M|    for (int y = starty4; y < endy4;
  ------------------
  |  Branch (228:27): [True: 2.26M, False: 98.1k]
  ------------------
  229|  2.26M|         y++, dst += 4 * PXSTRIDE(ls), lvl += b4_stride)
  ------------------
  |  |   53|  2.26M|#define PXSTRIDE(x) (x)
  ------------------
  230|  2.26M|    {
  231|  2.26M|        if (!have_top && !y) continue;
  ------------------
  |  Branch (231:13): [True: 333k, False: 1.93M]
  |  Branch (231:26): [True: 22.7k, False: 310k]
  ------------------
  232|  2.24M|        const uint32_t vmask[4] = {
  233|  2.24M|            mask[y][0][0] | ((unsigned) mask[y][0][1] << 16),
  234|  2.24M|            mask[y][1][0] | ((unsigned) mask[y][1][1] << 16),
  235|  2.24M|            mask[y][2][0] | ((unsigned) mask[y][2][1] << 16),
  236|  2.24M|            0,
  237|  2.24M|        };
  238|  2.24M|        loop_filter_sb(dst, ls, vmask,
  239|  2.24M|                       (const uint8_t(*)[4]) &lvl[0][1], b4_stride,
  240|  2.24M|                       &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  241|  2.24M|    }
  242|  98.1k|}
lf_apply_tmpl.c:filter_plane_rows_uv:
  291|  43.2k|{
  292|  43.2k|    const Dav1dDSPContext *const dsp = f->dsp;
  293|  43.2k|    const loopfilter_sb_fn loop_filter_sb = dsp->lf.loop_filter_sb[1][1];
  294|  43.2k|    ptrdiff_t off_l = 0;
  295|       |
  296|       |    //                                 block1
  297|       |    // filter edges between rows (e.g. ------)
  298|       |    //                                 block2
  299|   850k|    for (int y = starty4; y < endy4;
  ------------------
  |  Branch (299:27): [True: 806k, False: 43.2k]
  ------------------
  300|   806k|         y++, off_l += 4 * PXSTRIDE(ls), lvl += b4_stride)
  ------------------
  |  |   53|   806k|#define PXSTRIDE(x) (x)
  ------------------
  301|   806k|    {
  302|   806k|        if (!have_top && !y) continue;
  ------------------
  |  Branch (302:13): [True: 109k, False: 696k]
  |  Branch (302:26): [True: 5.54k, False: 104k]
  ------------------
  303|   801k|        const uint32_t vmask[3] = {
  304|   801k|            mask[y][0][0] | ((unsigned) mask[y][0][1] << (16 >> ss_hor)),
  305|   801k|            mask[y][1][0] | ((unsigned) mask[y][1][1] << (16 >> ss_hor)),
  306|   801k|            0,
  307|   801k|        };
  308|   801k|        loop_filter_sb(&u[off_l], ls, vmask,
  309|   801k|                       (const uint8_t(*)[4]) &lvl[0][2], b4_stride,
  310|   801k|                       &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  311|   801k|        loop_filter_sb(&v[off_l], ls, vmask,
  312|   801k|                       (const uint8_t(*)[4]) &lvl[0][3], b4_stride,
  313|   801k|                       &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  314|   801k|    }
  315|  43.2k|}
dav1d_copy_lpf_16bpc:
  106|  31.3k|{
  107|  31.3k|    const int have_tt = f->c->n_tc > 1;
  108|  31.3k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  109|  31.3k|    const int offset = 8 * !!sby;
  110|  31.3k|    const ptrdiff_t *const src_stride = f->cur.stride;
  111|  31.3k|    const ptrdiff_t *const lr_stride = f->sr_cur.p.stride;
  112|  31.3k|    const int tt_off = have_tt * sby * (4 << f->seq_hdr->sb128);
  113|  31.3k|    pixel *const dst[3] = {
  114|  31.3k|        f->lf.lr_lpf_line[0] + tt_off * PXSTRIDE(lr_stride[0]),
  115|  31.3k|        f->lf.lr_lpf_line[1] + tt_off * PXSTRIDE(lr_stride[1]),
  116|  31.3k|        f->lf.lr_lpf_line[2] + tt_off * PXSTRIDE(lr_stride[1])
  117|  31.3k|    };
  118|       |
  119|       |    // TODO Also check block level restore type to reduce copying.
  120|  31.3k|    const int restore_planes = f->lf.restore_planes;
  121|       |
  122|  31.3k|    if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (122:9): [True: 24.9k, False: 6.37k]
  |  Branch (122:29): [True: 5.26k, False: 1.11k]
  ------------------
  123|  30.2k|        const int h = f->cur.p.h;
  124|  30.2k|        const int w = f->bw << 2;
  125|  30.2k|        const int row_h = imin((sby + 1) << (6 + f->seq_hdr->sb128), h - 1);
  126|  30.2k|        const int y_stripe = (sby << (6 + f->seq_hdr->sb128)) - offset;
  127|  30.2k|        if (restore_planes & LR_RESTORE_Y || !resize)
  ------------------
  |  Branch (127:13): [True: 12.4k, False: 17.7k]
  |  Branch (127:46): [True: 15.9k, False: 1.77k]
  ------------------
  128|  28.4k|            backup_lpf(f, dst[0], lr_stride[0],
  129|  28.4k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  130|  28.4k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 1);
  131|  30.2k|        if (have_tt && resize) {
  ------------------
  |  Branch (131:13): [True: 0, False: 30.2k]
  |  Branch (131:24): [True: 0, False: 0]
  ------------------
  132|      0|            const ptrdiff_t cdef_off_y = sby * 4 * PXSTRIDE(src_stride[0]);
  133|      0|            backup_lpf(f, f->lf.cdef_lpf_line[0] + cdef_off_y, src_stride[0],
  134|      0|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  135|      0|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 0);
  136|      0|        }
  137|  30.2k|    }
  138|  31.3k|    if ((f->seq_hdr->cdef || restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) &&
  ------------------
  |  Branch (138:10): [True: 24.9k, False: 6.37k]
  |  Branch (138:30): [True: 2.45k, False: 3.92k]
  ------------------
  139|  27.4k|        f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400)
  ------------------
  |  Branch (139:9): [True: 16.8k, False: 10.6k]
  ------------------
  140|  16.8k|    {
  141|  16.8k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  142|  16.8k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  143|  16.8k|        const int h = (f->cur.p.h + ss_ver) >> ss_ver;
  144|  16.8k|        const int w = f->bw << (2 - ss_hor);
  145|  16.8k|        const int row_h = imin((sby + 1) << ((6 - ss_ver) + f->seq_hdr->sb128), h - 1);
  146|  16.8k|        const int offset_uv = offset >> ss_ver;
  147|  16.8k|        const int y_stripe = (sby << ((6 - ss_ver) + f->seq_hdr->sb128)) - offset_uv;
  148|  16.8k|        const ptrdiff_t cdef_off_uv = sby * 4 * PXSTRIDE(src_stride[1]);
  149|  16.8k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_U) {
  ------------------
  |  Branch (149:13): [True: 14.3k, False: 2.45k]
  |  Branch (149:33): [True: 2.10k, False: 344]
  ------------------
  150|  16.4k|            if (restore_planes & LR_RESTORE_U || !resize)
  ------------------
  |  Branch (150:17): [True: 5.30k, False: 11.1k]
  |  Branch (150:50): [True: 9.85k, False: 1.32k]
  ------------------
  151|  15.1k|                backup_lpf(f, dst[1], lr_stride[1],
  152|  15.1k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  153|  15.1k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  154|  15.1k|                           row_h, w, h, ss_hor, 1);
  155|  16.4k|            if (have_tt && resize)
  ------------------
  |  Branch (155:17): [True: 0, False: 16.4k]
  |  Branch (155:28): [True: 0, False: 0]
  ------------------
  156|      0|                backup_lpf(f, f->lf.cdef_lpf_line[1] + cdef_off_uv, src_stride[1],
  157|      0|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  158|      0|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  159|      0|                           row_h, w, h, ss_hor, 0);
  160|  16.4k|        }
  161|  16.8k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_V) {
  ------------------
  |  Branch (161:13): [True: 14.3k, False: 2.45k]
  |  Branch (161:33): [True: 1.50k, False: 948]
  ------------------
  162|  15.8k|            if (restore_planes & LR_RESTORE_V || !resize)
  ------------------
  |  Branch (162:17): [True: 4.56k, False: 11.3k]
  |  Branch (162:50): [True: 9.46k, False: 1.85k]
  ------------------
  163|  14.0k|                backup_lpf(f, dst[2], lr_stride[1],
  164|  14.0k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  165|  14.0k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  166|  14.0k|                           row_h, w, h, ss_hor, 1);
  167|  15.8k|            if (have_tt && resize)
  ------------------
  |  Branch (167:17): [True: 0, False: 15.8k]
  |  Branch (167:28): [True: 0, False: 0]
  ------------------
  168|      0|                backup_lpf(f, f->lf.cdef_lpf_line[2] + cdef_off_uv, src_stride[1],
  169|      0|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  170|      0|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  171|      0|                           row_h, w, h, ss_hor, 0);
  172|  15.8k|        }
  173|  16.8k|    }
  174|  31.3k|}
dav1d_loopfilter_sbrow_cols_16bpc:
  320|  31.7k|{
  321|  31.7k|    int x, have_left;
  322|       |    // Don't filter outside the frame
  323|  31.7k|    const int is_sb64 = !f->seq_hdr->sb128;
  324|  31.7k|    const int starty4 = (sby & is_sb64) << 4;
  325|  31.7k|    const int sbsz = 32 >> is_sb64;
  326|  31.7k|    const int sbl2 = 5 - is_sb64;
  327|  31.7k|    const int halign = (f->bh + 31) & ~31;
  328|  31.7k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  329|  31.7k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  330|  31.7k|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  331|  31.7k|    const unsigned vmax = 1U << vmask, hmax = 1U << hmask;
  332|  31.7k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  333|  31.7k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  334|       |
  335|       |    // fix lpf strength at tile col boundaries
  336|  31.7k|    const uint8_t *lpf_y = &f->lf.tx_lpf_right_edge[0][sby << sbl2];
  337|  31.7k|    const uint8_t *lpf_uv = &f->lf.tx_lpf_right_edge[1][sby << (sbl2 - ss_ver)];
  338|  32.9k|    for (int tile_col = 1;; tile_col++) {
  339|  32.9k|        x = f->frame_hdr->tiling.col_start_sb[tile_col];
  340|  32.9k|        if ((x << sbl2) >= f->bw) break;
  ------------------
  |  Branch (340:13): [True: 31.7k, False: 1.21k]
  ------------------
  341|  1.21k|        const int bx4 = x & is_sb64 ? 16 : 0, cbx4 = bx4 >> ss_hor;
  ------------------
  |  Branch (341:25): [True: 538, False: 672]
  ------------------
  342|  1.21k|        x >>= is_sb64;
  343|       |
  344|  1.21k|        uint16_t (*const y_hmask)[2] = lflvl[x].filter_y[0][bx4];
  345|  30.7k|        for (unsigned y = starty4, mask = 1 << y; y < endy4; y++, mask <<= 1) {
  ------------------
  |  Branch (345:51): [True: 29.5k, False: 1.21k]
  ------------------
  346|  29.5k|            const int sidx = mask >= 0x10000U;
  347|  29.5k|            const unsigned smask = mask >> (sidx << 4);
  348|  29.5k|            const int idx = 2 * !!(y_hmask[2][sidx] & smask) +
  349|  29.5k|                                !!(y_hmask[1][sidx] & smask);
  350|  29.5k|            y_hmask[2][sidx] &= ~smask;
  351|  29.5k|            y_hmask[1][sidx] &= ~smask;
  352|  29.5k|            y_hmask[0][sidx] &= ~smask;
  353|  29.5k|            y_hmask[imin(idx, lpf_y[y - starty4])][sidx] |= smask;
  354|  29.5k|        }
  355|       |
  356|  1.21k|        if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (356:13): [True: 346, False: 864]
  ------------------
  357|    346|            uint16_t (*const uv_hmask)[2] = lflvl[x].filter_uv[0][cbx4];
  358|  3.91k|            for (unsigned y = starty4 >> ss_ver, uv_mask = 1 << y; y < uv_endy4;
  ------------------
  |  Branch (358:68): [True: 3.56k, False: 346]
  ------------------
  359|  3.56k|                 y++, uv_mask <<= 1)
  360|  3.56k|            {
  361|  3.56k|                const int sidx = uv_mask >= vmax;
  362|  3.56k|                const unsigned smask = uv_mask >> (sidx << (4 - ss_ver));
  363|  3.56k|                const int idx = !!(uv_hmask[1][sidx] & smask);
  364|  3.56k|                uv_hmask[1][sidx] &= ~smask;
  365|  3.56k|                uv_hmask[0][sidx] &= ~smask;
  366|  3.56k|                uv_hmask[imin(idx, lpf_uv[y - (starty4 >> ss_ver)])][sidx] |= smask;
  367|  3.56k|            }
  368|    346|        }
  369|  1.21k|        lpf_y  += halign;
  370|  1.21k|        lpf_uv += halign >> ss_ver;
  371|  1.21k|    }
  372|       |
  373|       |    // fix lpf strength at tile row boundaries
  374|  31.7k|    if (start_of_tile_row) {
  ------------------
  |  Branch (374:9): [True: 575, False: 31.1k]
  ------------------
  375|    575|        const BlockContext *a;
  376|    575|        for (x = 0, a = &f->a[f->sb128w * (start_of_tile_row - 1)];
  377|  2.21k|             x < f->sb128w; x++, a++)
  ------------------
  |  Branch (377:14): [True: 1.63k, False: 575]
  ------------------
  378|  1.63k|        {
  379|  1.63k|            uint16_t (*const y_vmask)[2] = lflvl[x].filter_y[1][starty4];
  380|  1.63k|            const unsigned w = imin(32, f->w4 - (x << 5));
  381|  39.3k|            for (unsigned mask = 1, i = 0; i < w; mask <<= 1, i++) {
  ------------------
  |  Branch (381:44): [True: 37.7k, False: 1.63k]
  ------------------
  382|  37.7k|                const int sidx = mask >= 0x10000U;
  383|  37.7k|                const unsigned smask = mask >> (sidx << 4);
  384|  37.7k|                const int idx = 2 * !!(y_vmask[2][sidx] & smask) +
  385|  37.7k|                                    !!(y_vmask[1][sidx] & smask);
  386|  37.7k|                y_vmask[2][sidx] &= ~smask;
  387|  37.7k|                y_vmask[1][sidx] &= ~smask;
  388|  37.7k|                y_vmask[0][sidx] &= ~smask;
  389|  37.7k|                y_vmask[imin(idx, a->tx_lpf_y[i])][sidx] |= smask;
  390|  37.7k|            }
  391|       |
  392|  1.63k|            if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (392:17): [True: 1.10k, False: 534]
  ------------------
  393|  1.10k|                const unsigned cw = (w + ss_hor) >> ss_hor;
  394|  1.10k|                uint16_t (*const uv_vmask)[2] = lflvl[x].filter_uv[1][starty4 >> ss_ver];
  395|  25.9k|                for (unsigned uv_mask = 1, i = 0; i < cw; uv_mask <<= 1, i++) {
  ------------------
  |  Branch (395:51): [True: 24.8k, False: 1.10k]
  ------------------
  396|  24.8k|                    const int sidx = uv_mask >= hmax;
  397|  24.8k|                    const unsigned smask = uv_mask >> (sidx << (4 - ss_hor));
  398|  24.8k|                    const int idx = !!(uv_vmask[1][sidx] & smask);
  399|  24.8k|                    uv_vmask[1][sidx] &= ~smask;
  400|  24.8k|                    uv_vmask[0][sidx] &= ~smask;
  401|  24.8k|                    uv_vmask[imin(idx, a->tx_lpf_uv[i])][sidx] |= smask;
  402|  24.8k|                }
  403|  1.10k|            }
  404|  1.63k|        }
  405|    575|    }
  406|       |
  407|  31.7k|    pixel *ptr;
  408|  31.7k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  409|  82.5k|    for (ptr = p[0], have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (409:44): [True: 50.7k, False: 31.7k]
  ------------------
  410|  50.7k|         x++, have_left = 1, ptr += 128, level_ptr += 32)
  411|  50.7k|    {
  412|  50.7k|        filter_plane_cols_y(f, have_left, level_ptr, f->b4_stride,
  413|  50.7k|                            lflvl[x].filter_y[0], ptr, f->cur.stride[0],
  414|  50.7k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  415|  50.7k|    }
  416|       |
  417|  31.7k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (417:9): [True: 23.7k, False: 8.04k]
  |  Branch (417:46): [True: 14.7k, False: 8.98k]
  ------------------
  418|  14.7k|        return;
  419|       |
  420|  17.0k|    ptrdiff_t uv_off;
  421|  17.0k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  422|  38.0k|    for (uv_off = 0, have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (422:44): [True: 21.0k, False: 17.0k]
  ------------------
  423|  21.0k|         x++, have_left = 1, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  424|  21.0k|    {
  425|  21.0k|        filter_plane_cols_uv(f, have_left, level_ptr, f->b4_stride,
  426|  21.0k|                             lflvl[x].filter_uv[0],
  427|  21.0k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  428|  21.0k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  429|  21.0k|                             starty4 >> ss_ver, uv_endy4, ss_ver);
  430|  21.0k|    }
  431|  17.0k|}
dav1d_loopfilter_sbrow_rows_16bpc:
  436|  31.7k|{
  437|  31.7k|    int x;
  438|       |    // Don't filter outside the frame
  439|  31.7k|    const int have_top = sby > 0;
  440|  31.7k|    const int is_sb64 = !f->seq_hdr->sb128;
  441|  31.7k|    const int starty4 = (sby & is_sb64) << 4;
  442|  31.7k|    const int sbsz = 32 >> is_sb64;
  443|  31.7k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  444|  31.7k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  445|  31.7k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  446|  31.7k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  447|       |
  448|  31.7k|    pixel *ptr;
  449|  31.7k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  450|  82.5k|    for (ptr = p[0], x = 0; x < f->sb128w; x++, ptr += 128, level_ptr += 32) {
  ------------------
  |  Branch (450:29): [True: 50.7k, False: 31.7k]
  ------------------
  451|  50.7k|        filter_plane_rows_y(f, have_top, level_ptr, f->b4_stride,
  452|  50.7k|                            lflvl[x].filter_y[1], ptr, f->cur.stride[0],
  453|  50.7k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  454|  50.7k|    }
  455|       |
  456|  31.7k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (456:9): [True: 23.7k, False: 8.04k]
  |  Branch (456:46): [True: 14.7k, False: 8.98k]
  ------------------
  457|  14.7k|        return;
  458|       |
  459|  17.0k|    ptrdiff_t uv_off;
  460|  17.0k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  461|  38.0k|    for (uv_off = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (461:29): [True: 21.0k, False: 17.0k]
  ------------------
  462|  21.0k|         x++, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  463|  21.0k|    {
  464|  21.0k|        filter_plane_rows_uv(f, have_top, level_ptr, f->b4_stride,
  465|  21.0k|                             lflvl[x].filter_uv[1],
  466|  21.0k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  467|  21.0k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  468|  21.0k|                             starty4 >> ss_ver, uv_endy4, ss_hor);
  469|  21.0k|    }
  470|  17.0k|}

dav1d_create_lf_mask_intra:
  271|   503k|{
  272|   503k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  273|   503k|    const int bw4 = imin(iw - bx, b_dim[0]);
  274|   503k|    const int bh4 = imin(ih - by, b_dim[1]);
  275|   503k|    const int bx4 = bx & 31;
  276|   503k|    const int by4 = by & 31;
  277|   503k|    assert(bw4 >= 0 && bh4 >= 0);
  ------------------
  |  Branch (277:5): [True: 503k, False: 0]
  |  Branch (277:5): [True: 503k, False: 0]
  ------------------
  278|       |
  279|   503k|    if (bw4 && bh4) {
  ------------------
  |  Branch (279:9): [True: 500k, False: 2.16k]
  |  Branch (279:16): [True: 479k, False: 21.7k]
  ------------------
  280|   479k|        uint8_t (*level_cache_ptr)[4] = level_cache + by * b4_stride + bx;
  281|  2.96M|        for (int y = 0; y < bh4; y++) {
  ------------------
  |  Branch (281:25): [True: 2.48M, False: 479k]
  ------------------
  282|  28.7M|            for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (282:29): [True: 26.2M, False: 2.48M]
  ------------------
  283|  26.2M|                level_cache_ptr[x][0] = filter_level[0][0][0];
  284|  26.2M|                level_cache_ptr[x][1] = filter_level[1][0][0];
  285|  26.2M|            }
  286|  2.48M|            level_cache_ptr += b4_stride;
  287|  2.48M|        }
  288|       |
  289|   479k|        mask_edges_intra(lflvl->filter_y, by4, bx4, bw4, bh4, ytx, ay, ly);
  290|   479k|    }
  291|       |
  292|   503k|    if (!auv) return;
  ------------------
  |  Branch (292:9): [True: 146k, False: 356k]
  ------------------
  293|       |
  294|   356k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  295|   356k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  296|   356k|    const int cbw4 = imin(((iw + ss_hor) >> ss_hor) - (bx >> ss_hor),
  297|   356k|                          (b_dim[0] + ss_hor) >> ss_hor);
  298|   356k|    const int cbh4 = imin(((ih + ss_ver) >> ss_ver) - (by >> ss_ver),
  299|   356k|                          (b_dim[1] + ss_ver) >> ss_ver);
  300|   356k|    assert(cbw4 >= 0 && cbh4 >= 0);
  ------------------
  |  Branch (300:5): [True: 356k, False: 0]
  |  Branch (300:5): [True: 356k, False: 0]
  ------------------
  301|       |
  302|   356k|    if (!cbw4 || !cbh4) return;
  ------------------
  |  Branch (302:9): [True: 1.11k, False: 355k]
  |  Branch (302:18): [True: 13.6k, False: 341k]
  ------------------
  303|       |
  304|   341k|    const int cbx4 = bx4 >> ss_hor;
  305|   341k|    const int cby4 = by4 >> ss_ver;
  306|       |
  307|   341k|    uint8_t (*level_cache_ptr)[4] =
  308|   341k|        level_cache + (by >> ss_ver) * b4_stride + (bx >> ss_hor);
  309|  1.73M|    for (int y = 0; y < cbh4; y++) {
  ------------------
  |  Branch (309:21): [True: 1.39M, False: 341k]
  ------------------
  310|  11.6M|        for (int x = 0; x < cbw4; x++) {
  ------------------
  |  Branch (310:25): [True: 10.2M, False: 1.39M]
  ------------------
  311|  10.2M|            level_cache_ptr[x][2] = filter_level[2][0][0];
  312|  10.2M|            level_cache_ptr[x][3] = filter_level[3][0][0];
  313|  10.2M|        }
  314|  1.39M|        level_cache_ptr += b4_stride;
  315|  1.39M|    }
  316|       |
  317|   341k|    mask_edges_chroma(lflvl->filter_uv, cby4, cbx4, cbw4, cbh4, 0, uvtx,
  318|   341k|                      auv, luv, ss_hor, ss_ver);
  319|   341k|}
dav1d_create_lf_mask_inter:
  334|   774k|{
  335|   774k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  336|   774k|    const int bw4 = imin(iw - bx, b_dim[0]);
  337|   774k|    const int bh4 = imin(ih - by, b_dim[1]);
  338|   774k|    const int bx4 = bx & 31;
  339|   774k|    const int by4 = by & 31;
  340|   774k|    assert(bw4 >= 0 && bh4 >= 0);
  ------------------
  |  Branch (340:5): [True: 774k, False: 0]
  |  Branch (340:5): [True: 774k, False: 0]
  ------------------
  341|       |
  342|   774k|    if (bw4 && bh4) {
  ------------------
  |  Branch (342:9): [True: 770k, False: 3.74k]
  |  Branch (342:16): [True: 761k, False: 8.95k]
  ------------------
  343|   761k|        uint8_t (*level_cache_ptr)[4] = level_cache + by * b4_stride + bx;
  344|  5.28M|        for (int y = 0; y < bh4; y++) {
  ------------------
  |  Branch (344:25): [True: 4.52M, False: 761k]
  ------------------
  345|  60.5M|            for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (345:29): [True: 56.0M, False: 4.52M]
  ------------------
  346|  56.0M|                level_cache_ptr[x][0] = filter_level[0][0][0];
  347|  56.0M|                level_cache_ptr[x][1] = filter_level[1][0][0];
  348|  56.0M|            }
  349|  4.52M|            level_cache_ptr += b4_stride;
  350|  4.52M|        }
  351|       |
  352|   761k|        mask_edges_inter(lflvl->filter_y, by4, bx4, bw4, bh4, skip,
  353|   761k|                         max_ytx, tx_masks, ay, ly);
  354|   761k|    }
  355|       |
  356|   774k|    if (!auv) return;
  ------------------
  |  Branch (356:9): [True: 521k, False: 253k]
  ------------------
  357|       |
  358|   253k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  359|   253k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  360|   253k|    const int cbw4 = imin(((iw + ss_hor) >> ss_hor) - (bx >> ss_hor),
  361|   253k|                          (b_dim[0] + ss_hor) >> ss_hor);
  362|   253k|    const int cbh4 = imin(((ih + ss_ver) >> ss_ver) - (by >> ss_ver),
  363|   253k|                          (b_dim[1] + ss_ver) >> ss_ver);
  364|   253k|    assert(cbw4 >= 0 && cbh4 >= 0);
  ------------------
  |  Branch (364:5): [True: 253k, False: 0]
  |  Branch (364:5): [True: 253k, False: 0]
  ------------------
  365|       |
  366|   253k|    if (!cbw4 || !cbh4) return;
  ------------------
  |  Branch (366:9): [True: 773, False: 252k]
  |  Branch (366:18): [True: 1.00k, False: 251k]
  ------------------
  367|       |
  368|   251k|    const int cbx4 = bx4 >> ss_hor;
  369|   251k|    const int cby4 = by4 >> ss_ver;
  370|       |
  371|   251k|    uint8_t (*level_cache_ptr)[4] =
  372|   251k|        level_cache + (by >> ss_ver) * b4_stride + (bx >> ss_hor);
  373|  1.49M|    for (int y = 0; y < cbh4; y++) {
  ------------------
  |  Branch (373:21): [True: 1.24M, False: 251k]
  ------------------
  374|  15.0M|        for (int x = 0; x < cbw4; x++) {
  ------------------
  |  Branch (374:25): [True: 13.8M, False: 1.24M]
  ------------------
  375|  13.8M|            level_cache_ptr[x][2] = filter_level[2][0][0];
  376|  13.8M|            level_cache_ptr[x][3] = filter_level[3][0][0];
  377|  13.8M|        }
  378|  1.24M|        level_cache_ptr += b4_stride;
  379|  1.24M|    }
  380|       |
  381|   251k|    mask_edges_chroma(lflvl->filter_uv, cby4, cbx4, cbw4, cbh4, skip, uvtx,
  382|   251k|                      auv, luv, ss_hor, ss_ver);
  383|   251k|}
dav1d_calc_eih:
  385|  16.0k|void dav1d_calc_eih(Av1FilterLUT *const lim_lut, const int filter_sharpness) {
  386|       |    // set E/I/H values from loopfilter level
  387|  16.0k|    const int sharp = filter_sharpness;
  388|  1.04M|    for (int level = 0; level < 64; level++) {
  ------------------
  |  Branch (388:25): [True: 1.02M, False: 16.0k]
  ------------------
  389|  1.02M|        int limit = level;
  390|       |
  391|  1.02M|        if (sharp > 0) {
  ------------------
  |  Branch (391:13): [True: 519k, False: 509k]
  ------------------
  392|   519k|            limit >>= (sharp + 3) >> 2;
  393|   519k|            limit = imin(limit, 9 - sharp);
  394|   519k|        }
  395|  1.02M|        limit = imax(limit, 1);
  396|       |
  397|  1.02M|        lim_lut->i[level] = limit;
  398|  1.02M|        lim_lut->e[level] = 2 * (level + 2) + limit;
  399|  1.02M|    }
  400|  16.0k|    lim_lut->sharp[0] = (sharp + 3) >> 2;
  401|  16.0k|    lim_lut->sharp[1] = sharp ? 9 - sharp : 0xff;
  ------------------
  |  Branch (401:25): [True: 8.11k, False: 7.96k]
  ------------------
  402|  16.0k|}
dav1d_calc_lf_values:
  441|  60.8k|{
  442|  60.8k|    const int n_seg = hdr->segmentation.enabled ? 8 : 1;
  ------------------
  |  Branch (442:23): [True: 12.4k, False: 48.3k]
  ------------------
  443|       |
  444|  60.8k|    if (!hdr->loopfilter.level_y[0] && !hdr->loopfilter.level_y[1]) {
  ------------------
  |  Branch (444:9): [True: 40.3k, False: 20.5k]
  |  Branch (444:40): [True: 29.2k, False: 11.0k]
  ------------------
  445|  29.2k|        memset(lflvl_values, 0, sizeof(*lflvl_values) * n_seg);
  446|  29.2k|        return;
  447|  29.2k|    }
  448|       |
  449|  31.5k|    const Dav1dLoopfilterModeRefDeltas *const mr_deltas =
  450|  31.5k|        hdr->loopfilter.mode_ref_delta_enabled ?
  ------------------
  |  Branch (450:9): [True: 22.1k, False: 9.39k]
  ------------------
  451|  31.5k|        &hdr->loopfilter.mode_ref_deltas : NULL;
  452|   131k|    for (int s = 0; s < n_seg; s++) {
  ------------------
  |  Branch (452:21): [True: 100k, False: 31.5k]
  ------------------
  453|   100k|        const Dav1dSegmentationData *const segd =
  454|   100k|            hdr->segmentation.enabled ? &hdr->segmentation.seg_data.d[s] : NULL;
  ------------------
  |  Branch (454:13): [True: 78.5k, False: 21.7k]
  ------------------
  455|       |
  456|   100k|        calc_lf_value(lflvl_values[s][0], hdr->loopfilter.level_y[0],
  457|   100k|                      lf_delta[0], segd ? segd->delta_lf_y_v : 0, mr_deltas);
  ------------------
  |  Branch (457:36): [True: 78.5k, False: 21.7k]
  ------------------
  458|   100k|        calc_lf_value(lflvl_values[s][1], hdr->loopfilter.level_y[1],
  459|   100k|                      lf_delta[hdr->delta.lf.multi ? 1 : 0],
  ------------------
  |  Branch (459:32): [True: 34.7k, False: 65.6k]
  ------------------
  460|   100k|                      segd ? segd->delta_lf_y_h : 0, mr_deltas);
  ------------------
  |  Branch (460:23): [True: 78.5k, False: 21.7k]
  ------------------
  461|   100k|        calc_lf_value_chroma(lflvl_values[s][2], hdr->loopfilter.level_u,
  462|   100k|                             lf_delta[hdr->delta.lf.multi ? 2 : 0],
  ------------------
  |  Branch (462:39): [True: 34.7k, False: 65.6k]
  ------------------
  463|   100k|                             segd ? segd->delta_lf_u : 0, mr_deltas);
  ------------------
  |  Branch (463:30): [True: 78.5k, False: 21.7k]
  ------------------
  464|   100k|        calc_lf_value_chroma(lflvl_values[s][3], hdr->loopfilter.level_v,
  465|   100k|                             lf_delta[hdr->delta.lf.multi ? 3 : 0],
  ------------------
  |  Branch (465:39): [True: 34.7k, False: 65.6k]
  ------------------
  466|   100k|                             segd ? segd->delta_lf_v : 0, mr_deltas);
  ------------------
  |  Branch (466:30): [True: 78.5k, False: 21.7k]
  ------------------
  467|   100k|    }
  468|  31.5k|}
lf_mask.c:mask_edges_intra:
  152|   479k|{
  153|   479k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  154|   479k|    const int twl4 = t_dim->lw, thl4 = t_dim->lh;
  155|   479k|    const int twl4c = imin(2, twl4), thl4c = imin(2, thl4);
  156|   479k|    int y, x;
  157|       |
  158|       |    // left block edge
  159|   479k|    unsigned mask = 1U << by4;
  160|  2.96M|    for (y = 0; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (160:17): [True: 2.48M, False: 479k]
  ------------------
  161|  2.48M|        const int sidx = mask >= 0x10000;
  162|  2.48M|        const unsigned smask = mask >> (sidx << 4);
  163|  2.48M|        masks[0][bx4][imin(twl4c, l[y])][sidx] |= smask;
  164|  2.48M|    }
  165|       |
  166|       |    // top block edge
  167|  3.15M|    for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (167:35): [True: 2.67M, False: 479k]
  ------------------
  168|  2.67M|        const int sidx = mask >= 0x10000;
  169|  2.67M|        const unsigned smask = mask >> (sidx << 4);
  170|  2.67M|        masks[1][by4][imin(thl4c, a[x])][sidx] |= smask;
  171|  2.67M|    }
  172|       |
  173|       |    // inner (tx) left|right edges
  174|   479k|    const int hstep = t_dim->w;
  175|   479k|    unsigned t = 1U << by4;
  176|   479k|    unsigned inner = (unsigned) ((((uint64_t) t) << h4) - t);
  177|   479k|    unsigned inner1 = inner & 0xffff, inner2 = inner >> 16;
  178|   637k|    for (x = hstep; x < w4; x += hstep) {
  ------------------
  |  Branch (178:21): [True: 158k, False: 479k]
  ------------------
  179|   158k|        if (inner1) masks[0][bx4 + x][twl4c][0] |= inner1;
  ------------------
  |  Branch (179:13): [True: 137k, False: 21.2k]
  ------------------
  180|   158k|        if (inner2) masks[0][bx4 + x][twl4c][1] |= inner2;
  ------------------
  |  Branch (180:13): [True: 44.3k, False: 114k]
  ------------------
  181|   158k|    }
  182|       |
  183|       |    //            top
  184|       |    // inner (tx) --- edges
  185|       |    //           bottom
  186|   479k|    const int vstep = t_dim->h;
  187|   479k|    t = 1U << bx4;
  188|   479k|    inner = (unsigned) ((((uint64_t) t) << w4) - t);
  189|   479k|    inner1 = inner & 0xffff;
  190|   479k|    inner2 = inner >> 16;
  191|   610k|    for (y = vstep; y < h4; y += vstep) {
  ------------------
  |  Branch (191:21): [True: 131k, False: 479k]
  ------------------
  192|   131k|        if (inner1) masks[1][by4 + y][thl4c][0] |= inner1;
  ------------------
  |  Branch (192:13): [True: 81.6k, False: 49.5k]
  ------------------
  193|   131k|        if (inner2) masks[1][by4 + y][thl4c][1] |= inner2;
  ------------------
  |  Branch (193:13): [True: 67.9k, False: 63.2k]
  ------------------
  194|   131k|    }
  195|       |
  196|   479k|    dav1d_memset_likely_pow2(a, thl4c, w4);
  197|   479k|    dav1d_memset_likely_pow2(l, twl4c, h4);
  198|   479k|}
lf_mask.c:mask_edges_chroma:
  207|   593k|{
  208|   593k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  209|   593k|    const int twl4 = t_dim->lw, thl4 = t_dim->lh;
  210|   593k|    const int twl4c = !!twl4, thl4c = !!thl4;
  211|   593k|    int y, x;
  212|   593k|    const int vbits = 4 - ss_ver, hbits = 4 - ss_hor;
  213|   593k|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  214|   593k|    const unsigned vmax = 1 << vmask, hmax = 1 << hmask;
  215|       |
  216|       |    // left block edge
  217|   593k|    unsigned mask = 1U << cby4;
  218|  3.23M|    for (y = 0; y < ch4; y++, mask <<= 1) {
  ------------------
  |  Branch (218:17): [True: 2.63M, False: 593k]
  ------------------
  219|  2.63M|        const int sidx = mask >= vmax;
  220|  2.63M|        const unsigned smask = mask >> (sidx << vbits);
  221|  2.63M|        masks[0][cbx4][imin(twl4c, l[y])][sidx] |= smask;
  222|  2.63M|    }
  223|       |
  224|       |    // top block edge
  225|  3.35M|    for (x = 0, mask = 1U << cbx4; x < cw4; x++, mask <<= 1) {
  ------------------
  |  Branch (225:36): [True: 2.76M, False: 593k]
  ------------------
  226|  2.76M|        const int sidx = mask >= hmax;
  227|  2.76M|        const unsigned smask = mask >> (sidx << hbits);
  228|  2.76M|        masks[1][cby4][imin(thl4c, a[x])][sidx] |= smask;
  229|  2.76M|    }
  230|       |
  231|   593k|    if (!skip_inter) {
  ------------------
  |  Branch (231:9): [True: 447k, False: 146k]
  ------------------
  232|       |        // inner (tx) left|right edges
  233|   447k|        const int hstep = t_dim->w;
  234|   447k|        unsigned t = 1U << cby4;
  235|   447k|        unsigned inner = (unsigned) ((((uint64_t) t) << ch4) - t);
  236|   447k|        unsigned inner1 = inner & ((1 << vmask) - 1), inner2 = inner >> vmask;
  237|   487k|        for (x = hstep; x < cw4; x += hstep) {
  ------------------
  |  Branch (237:25): [True: 40.5k, False: 447k]
  ------------------
  238|  40.5k|            if (inner1) masks[0][cbx4 + x][twl4c][0] |= inner1;
  ------------------
  |  Branch (238:17): [True: 38.5k, False: 2.06k]
  ------------------
  239|  40.5k|            if (inner2) masks[0][cbx4 + x][twl4c][1] |= inner2;
  ------------------
  |  Branch (239:17): [True: 16.3k, False: 24.2k]
  ------------------
  240|  40.5k|        }
  241|       |
  242|       |        //            top
  243|       |        // inner (tx) --- edges
  244|       |        //           bottom
  245|   447k|        const int vstep = t_dim->h;
  246|   447k|        t = 1U << cbx4;
  247|   447k|        inner = (unsigned) ((((uint64_t) t) << cw4) - t);
  248|   447k|        inner1 = inner & ((1 << hmask) - 1), inner2 = inner >> hmask;
  249|   501k|        for (y = vstep; y < ch4; y += vstep) {
  ------------------
  |  Branch (249:25): [True: 54.5k, False: 447k]
  ------------------
  250|  54.5k|            if (inner1) masks[1][cby4 + y][thl4c][0] |= inner1;
  ------------------
  |  Branch (250:17): [True: 38.0k, False: 16.5k]
  ------------------
  251|  54.5k|            if (inner2) masks[1][cby4 + y][thl4c][1] |= inner2;
  ------------------
  |  Branch (251:17): [True: 33.8k, False: 20.7k]
  ------------------
  252|  54.5k|        }
  253|   447k|    }
  254|       |
  255|   593k|    dav1d_memset_likely_pow2(a, thl4c, cw4);
  256|   593k|    dav1d_memset_likely_pow2(l, twl4c, ch4);
  257|   593k|}
lf_mask.c:mask_edges_inter:
   85|   761k|{
   86|   761k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[max_tx];
   87|   761k|    int y, x;
   88|       |
   89|   761k|    ALIGN_STK_16(uint8_t, txa, 2 /* edge */, [2 /* txsz, step */][32 /* y */][32 /* x */]);
  ------------------
  |  |  100|   761k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   761k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
   90|  1.57M|    for (int y_off = 0, y = 0; y < h4; y += t_dim->h, y_off++)
  ------------------
  |  Branch (90:32): [True: 809k, False: 761k]
  ------------------
   91|  1.71M|        for (int x_off = 0, x = 0; x < w4; x += t_dim->w, x_off++)
  ------------------
  |  Branch (91:36): [True: 906k, False: 809k]
  ------------------
   92|   906k|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][y][x],
   93|   906k|                      max_tx, 0, y_off, x_off, tx_masks);
   94|       |
   95|       |    // left block edge
   96|   761k|    unsigned mask = 1U << by4;
   97|  5.28M|    for (y = 0; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (97:17): [True: 4.52M, False: 761k]
  ------------------
   98|  4.52M|        const int sidx = mask >= 0x10000;
   99|  4.52M|        const unsigned smask = mask >> (sidx << 4);
  100|  4.52M|        masks[0][bx4][imin(txa[0][0][y][0], l[y])][sidx] |= smask;
  101|  4.52M|    }
  102|       |
  103|       |    // top block edge
  104|  4.88M|    for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (104:35): [True: 4.11M, False: 761k]
  ------------------
  105|  4.11M|        const int sidx = mask >= 0x10000;
  106|  4.11M|        const unsigned smask = mask >> (sidx << 4);
  107|  4.11M|        masks[1][by4][imin(txa[1][0][0][x], a[x])][sidx] |= smask;
  108|  4.11M|    }
  109|       |
  110|   761k|    if (!skip) {
  ------------------
  |  Branch (110:9): [True: 214k, False: 547k]
  ------------------
  111|       |        // inner (tx) left|right edges
  112|  1.08M|        for (y = 0, mask = 1U << by4; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (112:39): [True: 875k, False: 214k]
  ------------------
  113|   875k|            const int sidx = mask >= 0x10000U;
  114|   875k|            const unsigned smask = mask >> (sidx << 4);
  115|   875k|            int ltx = txa[0][0][y][0];
  116|   875k|            int step = txa[0][1][y][0];
  117|  1.19M|            for (x = step; x < w4; x += step) {
  ------------------
  |  Branch (117:28): [True: 320k, False: 875k]
  ------------------
  118|   320k|                const int rtx = txa[0][0][y][x];
  119|   320k|                masks[0][bx4 + x][imin(rtx, ltx)][sidx] |= smask;
  120|   320k|                ltx = rtx;
  121|   320k|                step = txa[0][1][y][x];
  122|   320k|            }
  123|   875k|        }
  124|       |
  125|       |        //            top
  126|       |        // inner (tx) --- edges
  127|       |        //           bottom
  128|  1.16M|        for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (128:39): [True: 953k, False: 214k]
  ------------------
  129|   953k|            const int sidx = mask >= 0x10000U;
  130|   953k|            const unsigned smask = mask >> (sidx << 4);
  131|   953k|            int ttx = txa[1][0][0][x];
  132|   953k|            int step = txa[1][1][0][x];
  133|  1.26M|            for (y = step; y < h4; y += step) {
  ------------------
  |  Branch (133:28): [True: 311k, False: 953k]
  ------------------
  134|   311k|                const int btx = txa[1][0][y][x];
  135|   311k|                masks[1][by4 + y][imin(ttx, btx)][sidx] |= smask;
  136|   311k|                ttx = btx;
  137|   311k|                step = txa[1][1][y][x];
  138|   311k|            }
  139|   953k|        }
  140|   214k|    }
  141|       |
  142|  5.28M|    for (y = 0; y < h4; y++)
  ------------------
  |  Branch (142:17): [True: 4.52M, False: 761k]
  ------------------
  143|  4.52M|        l[y] = txa[0][0][y][w4 - 1];
  144|   761k|    memcpy(a, txa[1][0][h4 - 1], w4);
  145|   761k|}
lf_mask.c:decomp_tx:
   44|  1.08M|{
   45|  1.08M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[from];
   46|  1.08M|    const int is_split = (from == (int) TX_4X4 || depth > 1) ? 0 :
  ------------------
  |  Branch (46:27): [True: 152k, False: 932k]
  |  Branch (46:51): [True: 41.1k, False: 891k]
  ------------------
   47|  1.08M|        (tx_masks[depth] >> (y_off * 4 + x_off)) & 1;
   48|       |
   49|  1.08M|    if (is_split) {
  ------------------
  |  Branch (49:9): [True: 59.1k, False: 1.02M]
  ------------------
   50|  59.1k|        const enum RectTxfmSize sub = t_dim->sub;
   51|  59.1k|        const int htw4 = t_dim->w >> 1, hth4 = t_dim->h >> 1;
   52|       |
   53|  59.1k|        decomp_tx(txa, sub, depth + 1, y_off * 2 + 0, x_off * 2 + 0, tx_masks);
   54|  59.1k|        if (t_dim->w >= t_dim->h)
  ------------------
  |  Branch (54:13): [True: 47.3k, False: 11.8k]
  ------------------
   55|  47.3k|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][0][htw4],
   56|  47.3k|                      sub, depth + 1, y_off * 2 + 0, x_off * 2 + 1, tx_masks);
   57|  59.1k|        if (t_dim->h >= t_dim->w) {
  ------------------
  |  Branch (57:13): [True: 42.1k, False: 17.0k]
  ------------------
   58|  42.1k|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][hth4][0],
   59|  42.1k|                      sub, depth + 1, y_off * 2 + 1, x_off * 2 + 0, tx_masks);
   60|  42.1k|            if (t_dim->w >= t_dim->h)
  ------------------
  |  Branch (60:17): [True: 30.2k, False: 11.8k]
  ------------------
   61|  30.2k|                decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][hth4][htw4],
   62|  30.2k|                          sub, depth + 1, y_off * 2 + 1, x_off * 2 + 1, tx_masks);
   63|  42.1k|        }
   64|  1.02M|    } else {
   65|  1.02M|        const int lw = imin(2, t_dim->lw), lh = imin(2, t_dim->lh);
   66|       |
   67|  1.02M|#define set_ctx(rep_macro) \
   68|  1.02M|        for (int y = 0; y < t_dim->h; y++) { \
   69|  1.02M|            rep_macro(txa[0][0][y], 0, lw); \
   70|  1.02M|            rep_macro(txa[1][0][y], 0, lh); \
   71|  1.02M|            txa[0][1][y][0] = t_dim->w; \
   72|  1.02M|        }
   73|  1.02M|        case_set_upto16(t_dim->lw);
  ------------------
  |  |   80|  1.02M|    switch (var) { \
  |  |   81|   229k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |   68|   600k|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 371k, False: 229k]
  |  |  |  |  ------------------
  |  |  |  |   69|   371k|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|   371k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   371k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|   371k|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|   371k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   371k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|   371k|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|   371k|        }
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 229k, False: 796k]
  |  |  ------------------
  |  |   82|   277k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |   68|  1.05M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 779k, False: 277k]
  |  |  |  |  ------------------
  |  |  |  |   69|   779k|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|   779k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   779k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|   779k|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|   779k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   779k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|   779k|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|   779k|        }
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 277k, False: 748k]
  |  |  ------------------
  |  |   83|   246k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |   68|  1.31M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 1.06M, False: 246k]
  |  |  |  |  ------------------
  |  |  |  |   69|  1.06M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|  1.06M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.06M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  1.06M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|  1.06M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.06M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  1.06M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  1.06M|        }
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 246k, False: 779k]
  |  |  ------------------
  |  |   84|  73.6k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |   68|   536k|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 463k, False: 73.6k]
  |  |  |  |  ------------------
  |  |  |  |   69|   463k|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|   463k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   463k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|   463k|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|   463k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   463k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|   463k|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|   463k|        }
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 73.6k, False: 952k]
  |  |  ------------------
  |  |   85|   199k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |   68|  3.30M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 3.10M, False: 199k]
  |  |  |  |  ------------------
  |  |  |  |   69|  3.10M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.10M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  3.10M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  3.10M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  3.10M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 3.10M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  3.10M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.10M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  3.10M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  3.10M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  3.10M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 3.10M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  3.10M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  3.10M|        }
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 199k, False: 826k]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 1.02M]
  |  |  ------------------
  |  |   87|  1.02M|    }
  ------------------
  |  Branch (73:9): [Folded, False: 0]
  ------------------
   74|  1.02M|#undef set_ctx
   75|  1.02M|        dav1d_memset_pow2[t_dim->lw](txa[1][1][0], t_dim->h);
   76|  1.02M|    }
   77|  1.08M|}
lf_mask.c:calc_lf_value:
  408|   255k|{
  409|   255k|    const int base = iclip(iclip(base_lvl + lf_delta, 0, 63) + seg_delta, 0, 63);
  410|       |
  411|   255k|    if (!mr_delta) {
  ------------------
  |  Branch (411:9): [True: 78.2k, False: 177k]
  ------------------
  412|  78.2k|        memset(lflvl_values, base, sizeof(*lflvl_values) * 8);
  413|   177k|    } else {
  414|   177k|        const int sh = base >= 32;
  415|   177k|        lflvl_values[0][0] = lflvl_values[0][1] =
  416|   177k|            iclip(base + (mr_delta->ref_delta[0] * (1 << sh)), 0, 63);
  417|  1.41M|        for (int r = 1; r < 8; r++) {
  ------------------
  |  Branch (417:25): [True: 1.23M, False: 177k]
  ------------------
  418|  3.71M|            for (int m = 0; m < 2; m++) {
  ------------------
  |  Branch (418:29): [True: 2.47M, False: 1.23M]
  ------------------
  419|  2.47M|                const int delta =
  420|  2.47M|                    mr_delta->mode_delta[m] + mr_delta->ref_delta[r];
  421|  2.47M|                lflvl_values[r][m] = iclip(base + (delta * (1 << sh)), 0, 63);
  422|  2.47M|            }
  423|  1.23M|        }
  424|   177k|    }
  425|   255k|}
lf_mask.c:calc_lf_value_chroma:
  431|   200k|{
  432|   200k|    if (!base_lvl)
  ------------------
  |  Branch (432:9): [True: 145k, False: 54.6k]
  ------------------
  433|   145k|        memset(lflvl_values, 0, sizeof(*lflvl_values) * 8);
  434|  54.6k|    else
  435|  54.6k|        calc_lf_value(lflvl_values, base_lvl, lf_delta, seg_delta, mr_delta);
  436|   200k|}

dav1d_version:
   61|  10.2k|COLD const char *dav1d_version(void) {
   62|  10.2k|    return DAV1D_VERSION;
  ------------------
  |  |    2|  10.2k|#define DAV1D_VERSION "a34e068"
  ------------------
   63|  10.2k|}
dav1d_default_settings:
   71|  10.2k|COLD void dav1d_default_settings(Dav1dSettings *const s) {
   72|  10.2k|    s->n_threads = 0;
   73|  10.2k|    s->max_frame_delay = 0;
   74|  10.2k|    s->apply_grain = 1;
   75|  10.2k|    s->allocator.cookie = NULL;
   76|  10.2k|    s->allocator.alloc_picture_callback = dav1d_default_picture_alloc;
   77|  10.2k|    s->allocator.release_picture_callback = dav1d_default_picture_release;
   78|  10.2k|    s->logger.cookie = NULL;
   79|       |    s->logger.callback = dav1d_log_default_callback;
  ------------------
  |  |   43|  10.2k|#define dav1d_log_default_callback NULL
  ------------------
   80|  10.2k|    s->operating_point = 0;
   81|  10.2k|    s->all_layers = 1; // just until the tests are adjusted
   82|  10.2k|    s->frame_size_limit = 0;
   83|  10.2k|    s->strict_std_compliance = 0;
   84|  10.2k|    s->output_invisible_frames = 0;
   85|  10.2k|    s->inloop_filters = DAV1D_INLOOPFILTER_ALL;
   86|  10.2k|    s->decode_frame_type = DAV1D_DECODEFRAMETYPE_ALL;
   87|  10.2k|}
dav1d_open:
  140|  10.2k|COLD int dav1d_open(Dav1dContext **const c_out, const Dav1dSettings *const s) {
  141|  10.2k|    static pthread_once_t initted = PTHREAD_ONCE_INIT;
  142|  10.2k|    pthread_once(&initted, init_internal);
  143|       |
  144|  10.2k|    validate_input_or_ret(c_out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  10.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  145|  10.2k|    validate_input_or_ret(s != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  10.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  146|  10.2k|    validate_input_or_ret(s->n_threads >= 0 &&
  ------------------
  |  |   52|  20.4k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  147|  10.2k|                          s->n_threads <= DAV1D_MAX_THREADS, DAV1D_ERR(EINVAL));
  148|  10.2k|    validate_input_or_ret(s->max_frame_delay >= 0 &&
  ------------------
  |  |   52|  20.4k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  149|  10.2k|                          s->max_frame_delay <= DAV1D_MAX_FRAME_DELAY, DAV1D_ERR(EINVAL));
  150|  10.2k|    validate_input_or_ret(s->allocator.alloc_picture_callback != NULL,
  ------------------
  |  |   52|  10.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  151|  10.2k|                          DAV1D_ERR(EINVAL));
  152|  10.2k|    validate_input_or_ret(s->allocator.release_picture_callback != NULL,
  ------------------
  |  |   52|  10.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  153|  10.2k|                          DAV1D_ERR(EINVAL));
  154|  10.2k|    validate_input_or_ret(s->operating_point >= 0 &&
  ------------------
  |  |   52|  20.4k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  155|  10.2k|                          s->operating_point <= 31, DAV1D_ERR(EINVAL));
  156|  10.2k|    validate_input_or_ret(s->decode_frame_type >= DAV1D_DECODEFRAMETYPE_ALL &&
  ------------------
  |  |   52|  20.4k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  |  Branch (52:11): [True: 10.2k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  157|  10.2k|                          s->decode_frame_type <= DAV1D_DECODEFRAMETYPE_KEY, DAV1D_ERR(EINVAL));
  158|       |
  159|  10.2k|    pthread_attr_t thread_attr;
  160|  10.2k|    if (pthread_attr_init(&thread_attr)) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (160:9): [True: 0, False: 10.2k]
  ------------------
  161|  10.2k|    size_t stack_size = 1024 * 1024 + get_stack_size_internal(&thread_attr);
  162|       |
  163|  10.2k|    pthread_attr_setstacksize(&thread_attr, stack_size);
  164|       |
  165|  10.2k|    Dav1dContext *const c = *c_out = dav1d_alloc_aligned(ALLOC_COMMON_CTX, sizeof(*c), 64);
  ------------------
  |  |  134|  10.2k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  166|  10.2k|    if (!c) goto error;
  ------------------
  |  Branch (166:9): [True: 0, False: 10.2k]
  ------------------
  167|  10.2k|    memset(c, 0, sizeof(*c));
  168|       |
  169|  10.2k|    c->allocator = s->allocator;
  170|  10.2k|    c->logger = s->logger;
  171|  10.2k|    c->apply_grain = s->apply_grain;
  172|  10.2k|    c->operating_point = s->operating_point;
  173|  10.2k|    c->all_layers = s->all_layers;
  174|  10.2k|    c->frame_size_limit = s->frame_size_limit;
  175|  10.2k|    c->strict_std_compliance = s->strict_std_compliance;
  176|  10.2k|    c->output_invisible_frames = s->output_invisible_frames;
  177|  10.2k|    c->inloop_filters = s->inloop_filters;
  178|  10.2k|    c->decode_frame_type = s->decode_frame_type;
  179|       |
  180|  10.2k|    dav1d_data_props_set_defaults(&c->cached_error_props);
  181|       |
  182|  10.2k|    if (dav1d_mem_pool_init(ALLOC_OBU_HDR, &c->seq_hdr_pool) ||
  ------------------
  |  |  131|  20.4k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  183|  10.2k|        dav1d_mem_pool_init(ALLOC_OBU_HDR, &c->frame_hdr_pool) ||
  ------------------
  |  |  131|  20.4k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  184|  10.2k|        dav1d_mem_pool_init(ALLOC_SEGMAP, &c->segmap_pool) ||
  ------------------
  |  |  131|  20.4k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  185|  10.2k|        dav1d_mem_pool_init(ALLOC_REFMVS, &c->refmvs_pool) ||
  ------------------
  |  |  131|  20.4k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  186|  10.2k|        dav1d_mem_pool_init(ALLOC_PIC_CTX, &c->pic_ctx_pool) ||
  ------------------
  |  |  131|  20.4k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  187|  10.2k|        dav1d_mem_pool_init(ALLOC_CDF, &c->cdf_pool))
  ------------------
  |  |  131|  10.2k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  188|      0|    {
  189|      0|        goto error;
  190|      0|    }
  191|       |
  192|  10.2k|    if (c->allocator.alloc_picture_callback   == dav1d_default_picture_alloc &&
  ------------------
  |  Branch (192:9): [True: 10.2k, False: 0]
  ------------------
  193|  10.2k|        c->allocator.release_picture_callback == dav1d_default_picture_release)
  ------------------
  |  Branch (193:9): [True: 10.2k, False: 0]
  ------------------
  194|  10.2k|    {
  195|  10.2k|        if (c->allocator.cookie) goto error;
  ------------------
  |  Branch (195:13): [True: 0, False: 10.2k]
  ------------------
  196|  10.2k|        if (dav1d_mem_pool_init(ALLOC_PIC, &c->picture_pool)) goto error;
  ------------------
  |  |  131|  10.2k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 10.2k]
  |  |  ------------------
  ------------------
  197|  10.2k|        c->allocator.cookie = c->picture_pool;
  198|  10.2k|    } else if (c->allocator.alloc_picture_callback   == dav1d_default_picture_alloc ||
  ------------------
  |  Branch (198:16): [True: 0, False: 0]
  ------------------
  199|      0|               c->allocator.release_picture_callback == dav1d_default_picture_release)
  ------------------
  |  Branch (199:16): [True: 0, False: 0]
  ------------------
  200|      0|    {
  201|      0|        goto error;
  202|      0|    }
  203|       |
  204|       |    /* On 32-bit systems extremely large frame sizes can cause overflows in
  205|       |     * dav1d_decode_frame() malloc size calculations. Prevent that from occuring
  206|       |     * by enforcing a maximum frame size limit, chosen to roughly correspond to
  207|       |     * the largest size possible to decode without exhausting virtual memory. */
  208|  10.2k|    if (sizeof(size_t) < 8 && s->frame_size_limit - 1 >= 8192 * 8192) {
  ------------------
  |  Branch (208:9): [Folded, False: 10.2k]
  |  Branch (208:31): [True: 0, False: 0]
  ------------------
  209|      0|        c->frame_size_limit = 8192 * 8192;
  210|      0|        if (s->frame_size_limit)
  ------------------
  |  Branch (210:13): [True: 0, False: 0]
  ------------------
  211|      0|            dav1d_log(c, "Frame size limit reduced from %u to %u.\n",
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
  212|      0|                      s->frame_size_limit, c->frame_size_limit);
  213|      0|    }
  214|       |
  215|  10.2k|    c->flush = &c->flush_mem;
  216|  10.2k|    atomic_init(c->flush, 0);
  217|       |
  218|  10.2k|    get_num_threads(c, s, &c->n_tc, &c->n_fc);
  219|       |
  220|  10.2k|    c->fc = dav1d_alloc_aligned(ALLOC_THREAD_CTX, sizeof(*c->fc) * c->n_fc, 32);
  ------------------
  |  |  134|  10.2k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  221|  10.2k|    if (!c->fc) goto error;
  ------------------
  |  Branch (221:9): [True: 0, False: 10.2k]
  ------------------
  222|  10.2k|    memset(c->fc, 0, sizeof(*c->fc) * c->n_fc);
  223|       |
  224|  10.2k|    c->tc = dav1d_alloc_aligned(ALLOC_THREAD_CTX, sizeof(*c->tc) * c->n_tc, 64);
  ------------------
  |  |  134|  10.2k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  225|  10.2k|    if (!c->tc) goto error;
  ------------------
  |  Branch (225:9): [True: 0, False: 10.2k]
  ------------------
  226|  10.2k|    memset(c->tc, 0, sizeof(*c->tc) * c->n_tc);
  227|  10.2k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (227:9): [True: 0, False: 10.2k]
  ------------------
  228|      0|        if (pthread_mutex_init(&c->task_thread.lock, NULL)) goto error;
  ------------------
  |  Branch (228:13): [True: 0, False: 0]
  ------------------
  229|      0|        if (pthread_cond_init(&c->task_thread.cond, NULL)) {
  ------------------
  |  Branch (229:13): [True: 0, False: 0]
  ------------------
  230|      0|            pthread_mutex_destroy(&c->task_thread.lock);
  231|      0|            goto error;
  232|      0|        }
  233|      0|        if (pthread_cond_init(&c->task_thread.delayed_fg.cond, NULL)) {
  ------------------
  |  Branch (233:13): [True: 0, False: 0]
  ------------------
  234|      0|            pthread_cond_destroy(&c->task_thread.cond);
  235|      0|            pthread_mutex_destroy(&c->task_thread.lock);
  236|      0|            goto error;
  237|      0|        }
  238|      0|        c->task_thread.cur = c->n_fc;
  239|      0|        atomic_init(&c->task_thread.reset_task_cur, UINT_MAX);
  240|      0|        atomic_init(&c->task_thread.cond_signaled, 0);
  241|      0|        c->task_thread.inited = 1;
  242|      0|    }
  243|       |
  244|  10.2k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (244:9): [True: 0, False: 10.2k]
  ------------------
  245|      0|        const size_t out_delayed_sz = sizeof(*c->frame_thread.out_delayed) * c->n_fc;
  246|      0|        c->frame_thread.out_delayed =
  247|      0|            dav1d_malloc(ALLOC_THREAD_CTX, out_delayed_sz);
  ------------------
  |  |  132|      0|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
  248|      0|        if (!c->frame_thread.out_delayed) goto error;
  ------------------
  |  Branch (248:13): [True: 0, False: 0]
  ------------------
  249|      0|        memset(c->frame_thread.out_delayed, 0, out_delayed_sz);
  250|      0|    }
  251|  20.4k|    for (unsigned n = 0; n < c->n_fc; n++) {
  ------------------
  |  Branch (251:26): [True: 10.2k, False: 10.2k]
  ------------------
  252|  10.2k|        Dav1dFrameContext *const f = &c->fc[n];
  253|  10.2k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (253:13): [True: 0, False: 10.2k]
  ------------------
  254|      0|            if (pthread_mutex_init(&f->task_thread.lock, NULL)) goto error;
  ------------------
  |  Branch (254:17): [True: 0, False: 0]
  ------------------
  255|      0|            if (pthread_cond_init(&f->task_thread.cond, NULL)) {
  ------------------
  |  Branch (255:17): [True: 0, False: 0]
  ------------------
  256|      0|                pthread_mutex_destroy(&f->task_thread.lock);
  257|      0|                goto error;
  258|      0|            }
  259|      0|            if (pthread_mutex_init(&f->task_thread.pending_tasks.lock, NULL)) {
  ------------------
  |  Branch (259:17): [True: 0, False: 0]
  ------------------
  260|      0|                pthread_cond_destroy(&f->task_thread.cond);
  261|      0|                pthread_mutex_destroy(&f->task_thread.lock);
  262|      0|                goto error;
  263|      0|            }
  264|      0|        }
  265|  10.2k|        f->c = c;
  266|  10.2k|        f->task_thread.ttd = &c->task_thread;
  267|  10.2k|        f->lf.last_sharpness = -1;
  268|  10.2k|    }
  269|       |
  270|  20.4k|    for (unsigned m = 0; m < c->n_tc; m++) {
  ------------------
  |  Branch (270:26): [True: 10.2k, False: 10.2k]
  ------------------
  271|  10.2k|        Dav1dTaskContext *const t = &c->tc[m];
  272|  10.2k|        t->f = &c->fc[0];
  273|  10.2k|        t->task_thread.ttd = &c->task_thread;
  274|  10.2k|        t->c = c;
  275|  10.2k|        memset(t->cf_16bpc, 0, sizeof(t->cf_16bpc));
  276|  10.2k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (276:13): [True: 0, False: 10.2k]
  ------------------
  277|      0|            if (pthread_mutex_init(&t->task_thread.td.lock, NULL)) goto error;
  ------------------
  |  Branch (277:17): [True: 0, False: 0]
  ------------------
  278|      0|            if (pthread_cond_init(&t->task_thread.td.cond, NULL)) {
  ------------------
  |  Branch (278:17): [True: 0, False: 0]
  ------------------
  279|      0|                pthread_mutex_destroy(&t->task_thread.td.lock);
  280|      0|                goto error;
  281|      0|            }
  282|      0|            if (pthread_create(&t->task_thread.td.thread, &thread_attr, dav1d_worker_task, t)) {
  ------------------
  |  Branch (282:17): [True: 0, False: 0]
  ------------------
  283|      0|                pthread_cond_destroy(&t->task_thread.td.cond);
  284|      0|                pthread_mutex_destroy(&t->task_thread.td.lock);
  285|      0|                goto error;
  286|      0|            }
  287|      0|            t->task_thread.td.inited = 1;
  288|      0|        }
  289|  10.2k|    }
  290|  10.2k|    dav1d_pal_dsp_init(&c->pal_dsp);
  291|  10.2k|    dav1d_refmvs_dsp_init(&c->refmvs_dsp);
  292|       |
  293|  10.2k|    pthread_attr_destroy(&thread_attr);
  294|       |
  295|  10.2k|    return 0;
  296|       |
  297|      0|error:
  298|      0|    if (c) close_internal(c_out, 0);
  ------------------
  |  Branch (298:9): [True: 0, False: 0]
  ------------------
  299|      0|    pthread_attr_destroy(&thread_attr);
  300|      0|    return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  301|  10.2k|}
dav1d_send_data:
  439|  82.2k|{
  440|  82.2k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  82.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 82.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  441|  82.2k|    validate_input_or_ret(in != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  82.2k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 82.2k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  442|       |
  443|  82.2k|    if (in->data) {
  ------------------
  |  Branch (443:9): [True: 82.2k, False: 0]
  ------------------
  444|  82.2k|        validate_input_or_ret(in->sz > 0 && in->sz <= SIZE_MAX / 2, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   164k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 82.2k, False: 0]
  |  |  |  Branch (52:11): [True: 82.2k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  445|  82.2k|        c->drain = 0;
  446|  82.2k|    }
  447|  82.2k|    if (c->in.data)
  ------------------
  |  Branch (447:9): [True: 3.43k, False: 78.8k]
  ------------------
  448|  3.43k|        return DAV1D_ERR(EAGAIN);
  ------------------
  |  |   58|  3.43k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  449|  78.8k|    dav1d_data_ref(&c->in, in);
  450|       |
  451|  78.8k|    int res = gen_picture(c);
  452|  78.8k|    if (!res)
  ------------------
  |  Branch (452:9): [True: 25.1k, False: 53.6k]
  ------------------
  453|  25.1k|        dav1d_data_unref_internal(in);
  454|       |
  455|  78.8k|    return res;
  456|  82.2k|}
dav1d_get_picture:
  459|  40.0k|{
  460|  40.0k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  40.0k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 40.0k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  461|  40.0k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  40.0k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 40.0k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  462|       |
  463|  40.0k|    const int drain = c->drain;
  464|  40.0k|    c->drain = 1;
  465|       |
  466|  40.0k|    int res = gen_picture(c);
  467|  40.0k|    if (res < 0)
  ------------------
  |  Branch (467:9): [True: 1.97k, False: 38.0k]
  ------------------
  468|  1.97k|        return res;
  469|       |
  470|  38.0k|    if (c->cached_error) {
  ------------------
  |  Branch (470:9): [True: 0, False: 38.0k]
  ------------------
  471|      0|        const int res = c->cached_error;
  472|      0|        c->cached_error = 0;
  473|      0|        return res;
  474|      0|    }
  475|       |
  476|  38.0k|    if (output_picture_ready(c, c->n_fc == 1))
  ------------------
  |  Branch (476:9): [True: 22.2k, False: 15.8k]
  ------------------
  477|  22.2k|        return output_image(c, out);
  478|       |
  479|  15.8k|    if (c->n_fc > 1 && drain)
  ------------------
  |  Branch (479:9): [True: 0, False: 15.8k]
  |  Branch (479:24): [True: 0, False: 0]
  ------------------
  480|      0|        return drain_picture(c, out);
  481|       |
  482|  15.8k|    return DAV1D_ERR(EAGAIN);
  ------------------
  |  |   58|  15.8k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  483|  15.8k|}
dav1d_apply_grain:
  487|  4.31k|{
  488|  4.31k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  4.31k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 4.31k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  489|  4.31k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  4.31k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 4.31k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  490|  4.31k|    validate_input_or_ret(in != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  4.31k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 4.31k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  491|       |
  492|  4.31k|    if (!has_grain(in)) {
  ------------------
  |  Branch (492:9): [True: 0, False: 4.31k]
  ------------------
  493|      0|        dav1d_picture_ref(out, in);
  494|      0|        return 0;
  495|      0|    }
  496|       |
  497|  4.31k|    int res = dav1d_picture_alloc_copy(c, out, in->p.w, in);
  498|  4.31k|    if (res < 0) goto error;
  ------------------
  |  Branch (498:9): [True: 0, False: 4.31k]
  ------------------
  499|       |
  500|  4.31k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (500:9): [True: 0, False: 4.31k]
  ------------------
  501|      0|        dav1d_task_delayed_fg(c, out, in);
  502|  4.31k|    } else {
  503|  4.31k|        switch (out->p.bpc) {
  504|      0|#if CONFIG_8BPC
  505|  1.79k|        case 8:
  ------------------
  |  Branch (505:9): [True: 1.79k, False: 2.51k]
  ------------------
  506|  1.79k|            dav1d_apply_grain_8bpc(&c->dsp[0].fg, out, in);
  507|  1.79k|            break;
  508|      0|#endif
  509|      0|#if CONFIG_16BPC
  510|  1.72k|        case 10:
  ------------------
  |  Branch (510:9): [True: 1.72k, False: 2.58k]
  ------------------
  511|  2.51k|        case 12:
  ------------------
  |  Branch (511:9): [True: 791, False: 3.51k]
  ------------------
  512|  2.51k|            dav1d_apply_grain_16bpc(&c->dsp[(out->p.bpc >> 1) - 4].fg, out, in);
  513|  2.51k|            break;
  514|      0|#endif
  515|      0|        default: abort();
  ------------------
  |  Branch (515:9): [True: 0, False: 4.31k]
  ------------------
  516|  4.31k|        }
  517|  4.31k|    }
  518|       |
  519|  4.31k|    return 0;
  520|       |
  521|      0|error:
  522|      0|    dav1d_picture_unref_internal(out);
  523|      0|    return res;
  524|  4.31k|}
dav1d_flush:
  526|  10.2k|void dav1d_flush(Dav1dContext *const c) {
  527|  10.2k|    dav1d_data_unref_internal(&c->in);
  528|  10.2k|    if (c->out.p.frame_hdr)
  ------------------
  |  Branch (528:9): [True: 0, False: 10.2k]
  ------------------
  529|      0|        dav1d_thread_picture_unref(&c->out);
  530|  10.2k|    if (c->cache.p.frame_hdr)
  ------------------
  |  Branch (530:9): [True: 0, False: 10.2k]
  ------------------
  531|      0|        dav1d_thread_picture_unref(&c->cache);
  532|       |
  533|  10.2k|    c->drain = 0;
  534|  10.2k|    c->cached_error = 0;
  535|       |
  536|  92.0k|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (536:21): [True: 81.8k, False: 10.2k]
  ------------------
  537|  81.8k|        if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (537:13): [True: 21.9k, False: 59.8k]
  ------------------
  538|  21.9k|            dav1d_thread_picture_unref(&c->refs[i].p);
  539|  81.8k|        dav1d_ref_dec(&c->refs[i].segmap);
  540|  81.8k|        dav1d_ref_dec(&c->refs[i].refmvs);
  541|  81.8k|        dav1d_cdf_thread_unref(&c->cdf[i]);
  542|  81.8k|    }
  543|  10.2k|    c->frame_hdr = NULL;
  544|  10.2k|    c->seq_hdr = NULL;
  545|  10.2k|    dav1d_ref_dec(&c->seq_hdr_ref);
  546|       |
  547|  10.2k|    c->mastering_display = NULL;
  548|  10.2k|    c->content_light = NULL;
  549|  10.2k|    c->itut_t35 = NULL;
  550|  10.2k|    c->n_itut_t35 = 0;
  551|  10.2k|    dav1d_ref_dec(&c->mastering_display_ref);
  552|  10.2k|    dav1d_ref_dec(&c->content_light_ref);
  553|  10.2k|    dav1d_ref_dec(&c->itut_t35_ref);
  554|       |
  555|  10.2k|    dav1d_data_props_unref_internal(&c->cached_error_props);
  556|       |
  557|  10.2k|    if (c->n_fc == 1 && c->n_tc == 1) return;
  ------------------
  |  Branch (557:9): [True: 10.2k, False: 0]
  |  Branch (557:25): [True: 10.2k, False: 0]
  ------------------
  558|  10.2k|    atomic_store(c->flush, 1);
  559|       |
  560|      0|    if (c->n_tc > 1) {
  ------------------
  |  Branch (560:9): [True: 0, False: 0]
  ------------------
  561|      0|        pthread_mutex_lock(&c->task_thread.lock);
  562|       |        // stop running tasks in worker threads
  563|      0|        for (unsigned i = 0; i < c->n_tc; i++) {
  ------------------
  |  Branch (563:30): [True: 0, False: 0]
  ------------------
  564|      0|            Dav1dTaskContext *const tc = &c->tc[i];
  565|      0|            while (!tc->task_thread.flushed) {
  ------------------
  |  Branch (565:20): [True: 0, False: 0]
  ------------------
  566|      0|                pthread_cond_wait(&tc->task_thread.td.cond, &c->task_thread.lock);
  567|      0|            }
  568|      0|        }
  569|      0|        for (unsigned i = 0; i < c->n_fc; i++) {
  ------------------
  |  Branch (569:30): [True: 0, False: 0]
  ------------------
  570|      0|            c->fc[i].task_thread.task_head = NULL;
  571|      0|            c->fc[i].task_thread.task_tail = NULL;
  572|      0|            c->fc[i].task_thread.task_cur_prev = NULL;
  573|      0|            c->fc[i].task_thread.pending_tasks.head = NULL;
  574|      0|            c->fc[i].task_thread.pending_tasks.tail = NULL;
  575|      0|            atomic_init(&c->fc[i].task_thread.pending_tasks.merge, 0);
  576|      0|        }
  577|      0|        atomic_init(&c->task_thread.first, 0);
  578|      0|        c->task_thread.cur = c->n_fc;
  579|      0|        atomic_store(&c->task_thread.reset_task_cur, UINT_MAX);
  580|      0|        atomic_store(&c->task_thread.cond_signaled, 0);
  581|      0|        pthread_mutex_unlock(&c->task_thread.lock);
  582|      0|    }
  583|       |
  584|      0|    if (c->n_fc > 1) {
  ------------------
  |  Branch (584:9): [True: 0, False: 0]
  ------------------
  585|      0|        for (unsigned n = 0, next = c->frame_thread.next; n < c->n_fc; n++, next++) {
  ------------------
  |  Branch (585:59): [True: 0, False: 0]
  ------------------
  586|      0|            if (next == c->n_fc) next = 0;
  ------------------
  |  Branch (586:17): [True: 0, False: 0]
  ------------------
  587|      0|            Dav1dFrameContext *const f = &c->fc[next];
  588|      0|            dav1d_decode_frame_exit(f, -1);
  589|      0|            f->n_tile_data = 0;
  590|      0|            f->task_thread.retval = 0;
  591|      0|            f->task_thread.error = 0;
  592|      0|            Dav1dThreadPicture *out_delayed = &c->frame_thread.out_delayed[next];
  593|      0|            if (out_delayed->p.frame_hdr) {
  ------------------
  |  Branch (593:17): [True: 0, False: 0]
  ------------------
  594|      0|                dav1d_thread_picture_unref(out_delayed);
  595|      0|            }
  596|      0|        }
  597|      0|        c->frame_thread.next = 0;
  598|      0|    }
  599|       |    atomic_store(c->flush, 0);
  600|      0|}
dav1d_close:
  602|  10.2k|COLD void dav1d_close(Dav1dContext **const c_out) {
  603|  10.2k|    validate_input(c_out != NULL);
  ------------------
  |  |   59|  10.2k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  10.2k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 10.2k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  604|       |#if TRACK_HEAP_ALLOCATIONS
  605|       |    dav1d_log_alloc_stats(*c_out);
  606|       |#endif
  607|  10.2k|    close_internal(c_out, 1);
  608|  10.2k|}
dav1d_picture_unref:
  727|  22.2k|void dav1d_picture_unref(Dav1dPicture *const p) {
  728|  22.2k|    dav1d_picture_unref_internal(p);
  729|  22.2k|}
dav1d_data_create:
  731|  78.9k|uint8_t *dav1d_data_create(Dav1dData *const buf, const size_t sz) {
  732|  78.9k|    return dav1d_data_create_internal(buf, sz);
  733|  78.9k|}
dav1d_data_unref:
  756|  53.7k|void dav1d_data_unref(Dav1dData *const buf) {
  757|  53.7k|    dav1d_data_unref_internal(buf);
  758|  53.7k|}
lib.c:get_num_threads:
  111|  10.2k|{
  112|       |    /* ceil(sqrt(n)) */
  113|  10.2k|    static const uint8_t fc_lut[49] = {
  114|  10.2k|        1,                                     /*     1 */
  115|  10.2k|        2, 2, 2,                               /*  2- 4 */
  116|  10.2k|        3, 3, 3, 3, 3,                         /*  5- 9 */
  117|  10.2k|        4, 4, 4, 4, 4, 4, 4,                   /* 10-16 */
  118|  10.2k|        5, 5, 5, 5, 5, 5, 5, 5, 5,             /* 17-25 */
  119|  10.2k|        6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,       /* 26-36 */
  120|  10.2k|        7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, /* 37-49 */
  121|  10.2k|    };
  122|  10.2k|    *n_tc = s->n_threads ? s->n_threads :
  ------------------
  |  Branch (122:13): [True: 10.2k, False: 0]
  ------------------
  123|  10.2k|        iclip(dav1d_num_logical_processors(c), 1, DAV1D_MAX_THREADS);
  ------------------
  |  |   46|      0|#define DAV1D_MAX_THREADS 256
  ------------------
  124|  10.2k|    *n_fc = s->max_frame_delay ? umin(s->max_frame_delay, *n_tc) :
  ------------------
  |  Branch (124:13): [True: 10.2k, False: 0]
  ------------------
  125|  10.2k|            *n_tc < 50 ? fc_lut[*n_tc - 1] : 8; // min(8, ceil(sqrt(n)))
  ------------------
  |  Branch (125:13): [True: 0, False: 0]
  ------------------
  126|  10.2k|}
lib.c:init_internal:
   53|      1|static COLD void init_internal(void) {
   54|      1|    dav1d_init_cpu();
   55|      1|    dav1d_init_ii_wedge_masks();
   56|      1|    dav1d_init_intra_edge_tree();
   57|      1|    dav1d_init_qm_tables();
   58|      1|    dav1d_init_thread();
  ------------------
  |  |  144|      1|#define dav1d_init_thread() do {} while (0)
  |  |  ------------------
  |  |  |  Branch (144:42): [Folded, False: 1]
  |  |  ------------------
  ------------------
   59|      1|}
lib.c:get_stack_size_internal:
   93|  10.2k|static COLD size_t get_stack_size_internal(const pthread_attr_t *const thread_attr) {
   94|       |    /* glibc has an issue where the size of the TLS is subtracted from the stack
   95|       |     * size instead of allocated separately. As a result the specified stack
   96|       |     * size may be insufficient when used in an application with large amounts
   97|       |     * of TLS data. The following is a workaround to compensate for that.
   98|       |     * See https://sourceware.org/bugzilla/show_bug.cgi?id=11787 */
   99|  10.2k|    size_t (*const get_minstack)(const pthread_attr_t*) =
  100|  10.2k|        dlsym(RTLD_DEFAULT, "__pthread_get_minstack");
  101|  10.2k|    if (get_minstack)
  ------------------
  |  Branch (101:9): [True: 10.2k, False: 0]
  ------------------
  102|  10.2k|        return get_minstack(thread_attr) - PTHREAD_STACK_MIN;
  103|      0|    return 0;
  104|  10.2k|}
lib.c:gen_picture:
  413|   118k|{
  414|   118k|    Dav1dData *const in = &c->in;
  415|       |
  416|   118k|    if (output_picture_ready(c, 0))
  ------------------
  |  Branch (416:9): [True: 18.1k, False: 100k]
  ------------------
  417|  18.1k|        return 0;
  418|       |
  419|   137k|    while (in->sz > 0) {
  ------------------
  |  Branch (419:12): [True: 114k, False: 22.8k]
  ------------------
  420|   114k|        const ptrdiff_t res = dav1d_parse_obus(c, in);
  421|   114k|        if (res < 0) {
  ------------------
  |  Branch (421:13): [True: 55.6k, False: 58.7k]
  ------------------
  422|  55.6k|            dav1d_data_unref_internal(in);
  423|  58.7k|        } else {
  424|  58.7k|            assert((size_t)res <= in->sz);
  ------------------
  |  Branch (424:13): [True: 58.7k, False: 0]
  ------------------
  425|  58.7k|            in->sz -= res;
  426|  58.7k|            in->data += res;
  427|  58.7k|            if (!in->sz) dav1d_data_unref_internal(in);
  ------------------
  |  Branch (427:17): [True: 23.2k, False: 35.5k]
  ------------------
  428|  58.7k|        }
  429|   114k|        if (output_picture_ready(c, 0))
  ------------------
  |  Branch (429:13): [True: 22.2k, False: 92.1k]
  ------------------
  430|  22.2k|            break;
  431|  92.1k|        if (res < 0)
  ------------------
  |  Branch (431:13): [True: 55.6k, False: 36.5k]
  ------------------
  432|  55.6k|            return (int)res;
  433|  92.1k|    }
  434|       |
  435|  45.0k|    return 0;
  436|   100k|}
lib.c:output_picture_ready:
  332|   271k|static int output_picture_ready(Dav1dContext *const c, const int drain) {
  333|   271k|    if (c->cached_error) return 1;
  ------------------
  |  Branch (333:9): [True: 0, False: 271k]
  ------------------
  334|   271k|    if (!c->all_layers && c->max_spatial_id) {
  ------------------
  |  Branch (334:9): [True: 0, False: 271k]
  |  Branch (334:27): [True: 0, False: 0]
  ------------------
  335|      0|        if (c->out.p.data[0] && c->cache.p.data[0]) {
  ------------------
  |  Branch (335:13): [True: 0, False: 0]
  |  Branch (335:33): [True: 0, False: 0]
  ------------------
  336|      0|            if (c->max_spatial_id == c->cache.p.frame_hdr->spatial_id ||
  ------------------
  |  Branch (336:17): [True: 0, False: 0]
  ------------------
  337|      0|                c->out.flags & PICTURE_FLAG_NEW_TEMPORAL_UNIT)
  ------------------
  |  Branch (337:17): [True: 0, False: 0]
  ------------------
  338|      0|                return 1;
  339|      0|            dav1d_thread_picture_unref(&c->cache);
  340|      0|            dav1d_thread_picture_move_ref(&c->cache, &c->out);
  341|      0|            return 0;
  342|      0|        } else if (c->cache.p.data[0] && drain) {
  ------------------
  |  Branch (342:20): [True: 0, False: 0]
  |  Branch (342:42): [True: 0, False: 0]
  ------------------
  343|      0|            return 1;
  344|      0|        } else if (c->out.p.data[0]) {
  ------------------
  |  Branch (344:20): [True: 0, False: 0]
  ------------------
  345|      0|            dav1d_thread_picture_move_ref(&c->cache, &c->out);
  346|      0|            return 0;
  347|      0|        }
  348|      0|    }
  349|       |
  350|   271k|    return !!c->out.p.data[0];
  351|   271k|}
lib.c:output_image:
  312|  22.2k|{
  313|  22.2k|    int res = 0;
  314|       |
  315|  22.2k|    Dav1dThreadPicture *const in = (c->all_layers || !c->max_spatial_id)
  ------------------
  |  Branch (315:37): [True: 22.2k, False: 0]
  |  Branch (315:54): [True: 0, False: 0]
  ------------------
  316|  22.2k|                                   ? &c->out : &c->cache;
  317|  22.2k|    if (!c->apply_grain || !has_grain(&in->p)) {
  ------------------
  |  Branch (317:9): [True: 0, False: 22.2k]
  |  Branch (317:28): [True: 17.9k, False: 4.31k]
  ------------------
  318|  17.9k|        dav1d_picture_move_ref(out, &in->p);
  319|  17.9k|        dav1d_thread_picture_unref(in);
  320|  17.9k|        goto end;
  321|  17.9k|    }
  322|       |
  323|  4.31k|    res = dav1d_apply_grain(c, out, &in->p);
  324|  4.31k|    dav1d_thread_picture_unref(in);
  325|  22.2k|end:
  326|  22.2k|    if (!c->all_layers && c->max_spatial_id && c->out.p.data[0]) {
  ------------------
  |  Branch (326:9): [True: 0, False: 22.2k]
  |  Branch (326:27): [True: 0, False: 0]
  |  Branch (326:48): [True: 0, False: 0]
  ------------------
  327|      0|        dav1d_thread_picture_move_ref(in, &c->out);
  328|      0|    }
  329|  22.2k|    return res;
  330|  4.31k|}
lib.c:has_grain:
  304|  26.5k|{
  305|  26.5k|    const Dav1dFilmGrainData *fgdata = &pic->frame_hdr->film_grain.data;
  306|  26.5k|    return fgdata->num_y_points || fgdata->num_uv_points[0] ||
  ------------------
  |  Branch (306:12): [True: 5.78k, False: 20.8k]
  |  Branch (306:36): [True: 1.42k, False: 19.3k]
  ------------------
  307|  19.3k|           fgdata->num_uv_points[1] || (fgdata->clip_to_restricted_range &&
  ------------------
  |  Branch (307:12): [True: 542, False: 18.8k]
  |  Branch (307:41): [True: 1.12k, False: 17.7k]
  ------------------
  308|  1.12k|                                        fgdata->chroma_scaling_from_luma);
  ------------------
  |  Branch (308:41): [True: 876, False: 245]
  ------------------
  309|  26.5k|}
lib.c:close_internal:
  610|  10.2k|static COLD void close_internal(Dav1dContext **const c_out, int flush) {
  611|  10.2k|    Dav1dContext *const c = *c_out;
  612|  10.2k|    if (!c) return;
  ------------------
  |  Branch (612:9): [True: 0, False: 10.2k]
  ------------------
  613|       |
  614|  10.2k|    if (flush) dav1d_flush(c);
  ------------------
  |  Branch (614:9): [True: 10.2k, False: 0]
  ------------------
  615|       |
  616|  10.2k|    if (c->tc) {
  ------------------
  |  Branch (616:9): [True: 10.2k, False: 0]
  ------------------
  617|  10.2k|        struct TaskThreadData *ttd = &c->task_thread;
  618|  10.2k|        if (ttd->inited) {
  ------------------
  |  Branch (618:13): [True: 0, False: 10.2k]
  ------------------
  619|      0|            pthread_mutex_lock(&ttd->lock);
  620|      0|            for (unsigned n = 0; n < c->n_tc && c->tc[n].task_thread.td.inited; n++)
  ------------------
  |  Branch (620:34): [True: 0, False: 0]
  |  Branch (620:49): [True: 0, False: 0]
  ------------------
  621|      0|                c->tc[n].task_thread.die = 1;
  622|      0|            pthread_cond_broadcast(&ttd->cond);
  623|      0|            pthread_mutex_unlock(&ttd->lock);
  624|      0|            for (unsigned n = 0; n < c->n_tc; n++) {
  ------------------
  |  Branch (624:34): [True: 0, False: 0]
  ------------------
  625|      0|                Dav1dTaskContext *const pf = &c->tc[n];
  626|      0|                if (!pf->task_thread.td.inited) break;
  ------------------
  |  Branch (626:21): [True: 0, False: 0]
  ------------------
  627|      0|                pthread_join(pf->task_thread.td.thread, NULL);
  628|      0|                pthread_cond_destroy(&pf->task_thread.td.cond);
  629|      0|                pthread_mutex_destroy(&pf->task_thread.td.lock);
  630|      0|            }
  631|      0|            pthread_cond_destroy(&ttd->delayed_fg.cond);
  632|      0|            pthread_cond_destroy(&ttd->cond);
  633|      0|            pthread_mutex_destroy(&ttd->lock);
  634|      0|        }
  635|  10.2k|        dav1d_free_aligned(c->tc);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  636|  10.2k|    }
  637|       |
  638|  20.4k|    for (unsigned n = 0; c->fc && n < c->n_fc; n++) {
  ------------------
  |  Branch (638:26): [True: 20.4k, False: 0]
  |  Branch (638:35): [True: 10.2k, False: 10.2k]
  ------------------
  639|  10.2k|        Dav1dFrameContext *const f = &c->fc[n];
  640|       |
  641|       |        // clean-up threading stuff
  642|  10.2k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (642:13): [True: 0, False: 10.2k]
  ------------------
  643|      0|            dav1d_free(f->tile_thread.lowest_pixel_mem);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  644|      0|            dav1d_free(f->frame_thread.b);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  645|      0|            dav1d_free_aligned(f->frame_thread.cbi);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  646|      0|            dav1d_free_aligned(f->frame_thread.pal_idx);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  647|      0|            dav1d_free_aligned(f->frame_thread.cf);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  648|      0|            dav1d_free(f->frame_thread.tile_start_off);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  649|      0|            dav1d_free_aligned(f->frame_thread.pal);
  ------------------
  |  |  136|      0|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  650|      0|        }
  651|  10.2k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (651:13): [True: 0, False: 10.2k]
  ------------------
  652|      0|            pthread_mutex_destroy(&f->task_thread.pending_tasks.lock);
  653|      0|            pthread_cond_destroy(&f->task_thread.cond);
  654|      0|            pthread_mutex_destroy(&f->task_thread.lock);
  655|      0|        }
  656|  10.2k|        dav1d_free(f->frame_thread.frame_progress);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  657|  10.2k|        dav1d_free(f->task_thread.tasks);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  658|  10.2k|        dav1d_free(f->task_thread.tile_tasks[0]);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  659|  10.2k|        dav1d_free_aligned(f->ts);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  660|  10.2k|        dav1d_free_aligned(f->ipred_edge[0]);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  661|  10.2k|        dav1d_free(f->a);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  662|  10.2k|        dav1d_free(f->tile);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  663|  10.2k|        dav1d_free(f->lf.mask);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  664|  10.2k|        dav1d_free(f->lf.level);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  665|  10.2k|        dav1d_free(f->lf.lr_mask);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  666|  10.2k|        dav1d_free(f->lf.tx_lpf_right_edge[0]);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  667|  10.2k|        dav1d_free(f->lf.start_of_tile_row);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  668|  10.2k|        dav1d_free_aligned(f->rf.r);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  669|  10.2k|        dav1d_free_aligned(f->lf.cdef_line_buf);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  670|  10.2k|        dav1d_free_aligned(f->lf.lr_line_buf);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  671|  10.2k|    }
  672|  10.2k|    dav1d_free_aligned(c->fc);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  673|  10.2k|    if (c->n_fc > 1 && c->frame_thread.out_delayed) {
  ------------------
  |  Branch (673:9): [True: 0, False: 10.2k]
  |  Branch (673:24): [True: 0, False: 0]
  ------------------
  674|      0|        for (unsigned n = 0; n < c->n_fc; n++)
  ------------------
  |  Branch (674:30): [True: 0, False: 0]
  ------------------
  675|      0|            if (c->frame_thread.out_delayed[n].p.frame_hdr)
  ------------------
  |  Branch (675:17): [True: 0, False: 0]
  ------------------
  676|      0|                dav1d_thread_picture_unref(&c->frame_thread.out_delayed[n]);
  677|      0|        dav1d_free(c->frame_thread.out_delayed);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  678|      0|    }
  679|  10.3k|    for (int n = 0; n < c->n_tile_data; n++)
  ------------------
  |  Branch (679:21): [True: 127, False: 10.2k]
  ------------------
  680|    127|        dav1d_data_unref_internal(&c->tile[n].data);
  681|  10.2k|    dav1d_free(c->tile);
  ------------------
  |  |  135|  10.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
  682|  92.0k|    for (int n = 0; n < 8; n++) {
  ------------------
  |  Branch (682:21): [True: 81.8k, False: 10.2k]
  ------------------
  683|  81.8k|        dav1d_cdf_thread_unref(&c->cdf[n]);
  684|  81.8k|        if (c->refs[n].p.p.frame_hdr)
  ------------------
  |  Branch (684:13): [True: 0, False: 81.8k]
  ------------------
  685|      0|            dav1d_thread_picture_unref(&c->refs[n].p);
  686|  81.8k|        dav1d_ref_dec(&c->refs[n].refmvs);
  687|  81.8k|        dav1d_ref_dec(&c->refs[n].segmap);
  688|  81.8k|    }
  689|  10.2k|    dav1d_ref_dec(&c->seq_hdr_ref);
  690|  10.2k|    dav1d_ref_dec(&c->frame_hdr_ref);
  691|       |
  692|  10.2k|    dav1d_ref_dec(&c->mastering_display_ref);
  693|  10.2k|    dav1d_ref_dec(&c->content_light_ref);
  694|  10.2k|    dav1d_ref_dec(&c->itut_t35_ref);
  695|       |
  696|  10.2k|    dav1d_mem_pool_end(c->seq_hdr_pool);
  697|  10.2k|    dav1d_mem_pool_end(c->frame_hdr_pool);
  698|  10.2k|    dav1d_mem_pool_end(c->segmap_pool);
  699|  10.2k|    dav1d_mem_pool_end(c->refmvs_pool);
  700|  10.2k|    dav1d_mem_pool_end(c->cdf_pool);
  701|  10.2k|    dav1d_mem_pool_end(c->picture_pool);
  702|  10.2k|    dav1d_mem_pool_end(c->pic_ctx_pool);
  703|       |
  704|  10.2k|    dav1d_freep_aligned(c_out);
  705|  10.2k|}

dav1d_loop_filter_dsp_init_8bpc:
  259|  3.66k|COLD void bitfn(dav1d_loop_filter_dsp_init)(Dav1dLoopFilterDSPContext *const c) {
  260|  3.66k|    c->loop_filter_sb[0][0] = loop_filter_h_sb128y_c;
  261|  3.66k|    c->loop_filter_sb[0][1] = loop_filter_v_sb128y_c;
  262|  3.66k|    c->loop_filter_sb[1][0] = loop_filter_h_sb128uv_c;
  263|  3.66k|    c->loop_filter_sb[1][1] = loop_filter_v_sb128uv_c;
  264|       |
  265|  3.66k|#if HAVE_ASM
  266|       |#if ARCH_AARCH64 || ARCH_ARM
  267|       |    loop_filter_dsp_init_arm(c);
  268|       |#elif ARCH_LOONGARCH64
  269|       |    loop_filter_dsp_init_loongarch(c);
  270|       |#elif ARCH_PPC64LE
  271|       |    loop_filter_dsp_init_ppc(c);
  272|       |#elif ARCH_X86
  273|       |    loop_filter_dsp_init_x86(c);
  274|  3.66k|#endif
  275|  3.66k|#endif
  276|  3.66k|}
dav1d_loop_filter_dsp_init_16bpc:
  259|  4.99k|COLD void bitfn(dav1d_loop_filter_dsp_init)(Dav1dLoopFilterDSPContext *const c) {
  260|  4.99k|    c->loop_filter_sb[0][0] = loop_filter_h_sb128y_c;
  261|  4.99k|    c->loop_filter_sb[0][1] = loop_filter_v_sb128y_c;
  262|  4.99k|    c->loop_filter_sb[1][0] = loop_filter_h_sb128uv_c;
  263|  4.99k|    c->loop_filter_sb[1][1] = loop_filter_v_sb128uv_c;
  264|       |
  265|  4.99k|#if HAVE_ASM
  266|       |#if ARCH_AARCH64 || ARCH_ARM
  267|       |    loop_filter_dsp_init_arm(c);
  268|       |#elif ARCH_LOONGARCH64
  269|       |    loop_filter_dsp_init_loongarch(c);
  270|       |#elif ARCH_PPC64LE
  271|       |    loop_filter_dsp_init_ppc(c);
  272|       |#elif ARCH_X86
  273|       |    loop_filter_dsp_init_x86(c);
  274|  4.99k|#endif
  275|  4.99k|#endif
  276|  4.99k|}

dav1d_loop_restoration_dsp_init_8bpc:
 1367|  3.66k|{
 1368|  3.66k|    c->wiener[0] = c->wiener[1] = wiener_c;
 1369|  3.66k|    c->sgr[0] = sgr_5x5_c;
 1370|  3.66k|    c->sgr[1] = sgr_3x3_c;
 1371|  3.66k|    c->sgr[2] = sgr_mix_c;
 1372|       |
 1373|  3.66k|#if HAVE_ASM
 1374|       |#if ARCH_AARCH64 || ARCH_ARM
 1375|       |    loop_restoration_dsp_init_arm(c, bpc);
 1376|       |#elif ARCH_LOONGARCH64
 1377|       |    loop_restoration_dsp_init_loongarch(c, bpc);
 1378|       |#elif ARCH_PPC64LE
 1379|       |    loop_restoration_dsp_init_ppc(c, bpc);
 1380|       |#elif ARCH_X86
 1381|       |    loop_restoration_dsp_init_x86(c, bpc);
 1382|  3.66k|#endif
 1383|  3.66k|#endif
 1384|  3.66k|}
looprestoration_tmpl.c:sgr_5x5_c:
  830|  4.80k|{
  831|  4.80k|    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  4.80k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  4.80k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  832|  4.80k|    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  4.80k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  4.80k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  833|  4.80k|    int32_t *sumsq_ptrs[5], *sumsq_rows[5];
  834|  4.80k|    coef *sum_ptrs[5], *sum_rows[5];
  835|  28.8k|    for (int i = 0; i < 5; i++) {
  ------------------
  |  Branch (835:21): [True: 24.0k, False: 4.80k]
  ------------------
  836|  24.0k|        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  24.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  837|  24.0k|        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  24.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  838|  24.0k|    }
  839|       |
  840|  4.80k|    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  4.80k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  4.80k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  841|  4.80k|    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  4.80k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  4.80k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  842|  4.80k|    int32_t *A_ptrs[2];
  843|  4.80k|    coef *B_ptrs[2];
  844|  14.4k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (844:21): [True: 9.61k, False: 4.80k]
  ------------------
  845|  9.61k|        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.61k|#define BUF_STRIDE (384 + 16)
  ------------------
  846|  9.61k|        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.61k|#define BUF_STRIDE (384 + 16)
  ------------------
  847|  9.61k|    }
  848|  4.80k|    const pixel *src = dst;
  849|  4.80k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  4.80k|#define PXSTRIDE(x) (x)
  ------------------
  850|       |
  851|  4.80k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (851:9): [True: 2.19k, False: 2.61k]
  ------------------
  852|  2.19k|        sumsq_ptrs[0] = sumsq_rows[0];
  853|  2.19k|        sumsq_ptrs[1] = sumsq_rows[0];
  854|  2.19k|        sumsq_ptrs[2] = sumsq_rows[1];
  855|  2.19k|        sumsq_ptrs[3] = sumsq_rows[2];
  856|  2.19k|        sumsq_ptrs[4] = sumsq_rows[3];
  857|  2.19k|        sum_ptrs[0] = sum_rows[0];
  858|  2.19k|        sum_ptrs[1] = sum_rows[0];
  859|  2.19k|        sum_ptrs[2] = sum_rows[1];
  860|  2.19k|        sum_ptrs[3] = sum_rows[2];
  861|  2.19k|        sum_ptrs[4] = sum_rows[3];
  862|       |
  863|  2.19k|        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
  864|  2.19k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.19k|#define PXSTRIDE(x) (x)
  ------------------
  865|  2.19k|        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
  866|       |
  867|  2.19k|        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
  868|  2.19k|        left++;
  869|  2.19k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.19k|#define PXSTRIDE(x) (x)
  ------------------
  870|       |
  871|  2.19k|        if (--h <= 0)
  ------------------
  |  Branch (871:13): [True: 302, False: 1.88k]
  ------------------
  872|    302|            goto vert_1;
  873|       |
  874|  1.88k|        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
  875|  1.88k|        left++;
  876|  1.88k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.88k|#define PXSTRIDE(x) (x)
  ------------------
  877|  1.88k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  878|  1.88k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.88k|#define BITDEPTH_MAX 0xff
  ------------------
  879|  1.88k|        rotate(A_ptrs, B_ptrs, 2);
  880|       |
  881|  1.88k|        if (--h <= 0)
  ------------------
  |  Branch (881:13): [True: 284, False: 1.60k]
  ------------------
  882|    284|            goto vert_2;
  883|       |
  884|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
  885|       |        // one of them to point at the previously unused rows[4].
  886|  1.60k|        sumsq_ptrs[3] = sumsq_rows[4];
  887|  1.60k|        sum_ptrs[3] = sum_rows[4];
  888|  2.61k|    } else {
  889|  2.61k|        sumsq_ptrs[0] = sumsq_rows[0];
  890|  2.61k|        sumsq_ptrs[1] = sumsq_rows[0];
  891|  2.61k|        sumsq_ptrs[2] = sumsq_rows[0];
  892|  2.61k|        sumsq_ptrs[3] = sumsq_rows[0];
  893|  2.61k|        sumsq_ptrs[4] = sumsq_rows[0];
  894|  2.61k|        sum_ptrs[0] = sum_rows[0];
  895|  2.61k|        sum_ptrs[1] = sum_rows[0];
  896|  2.61k|        sum_ptrs[2] = sum_rows[0];
  897|  2.61k|        sum_ptrs[3] = sum_rows[0];
  898|  2.61k|        sum_ptrs[4] = sum_rows[0];
  899|       |
  900|  2.61k|        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
  901|  2.61k|        left++;
  902|  2.61k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.61k|#define PXSTRIDE(x) (x)
  ------------------
  903|       |
  904|  2.61k|        if (--h <= 0)
  ------------------
  |  Branch (904:13): [True: 336, False: 2.27k]
  ------------------
  905|    336|            goto vert_1;
  906|       |
  907|  2.27k|        sumsq_ptrs[4] = sumsq_rows[1];
  908|  2.27k|        sum_ptrs[4] = sum_rows[1];
  909|       |
  910|  2.27k|        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], left, src, w, edges);
  911|  2.27k|        left++;
  912|  2.27k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.27k|#define PXSTRIDE(x) (x)
  ------------------
  913|       |
  914|  2.27k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  915|  2.27k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.27k|#define BITDEPTH_MAX 0xff
  ------------------
  916|  2.27k|        rotate(A_ptrs, B_ptrs, 2);
  917|       |
  918|  2.27k|        if (--h <= 0)
  ------------------
  |  Branch (918:13): [True: 322, False: 1.95k]
  ------------------
  919|    322|            goto vert_2;
  920|       |
  921|  1.95k|        sumsq_ptrs[3] = sumsq_rows[2];
  922|  1.95k|        sumsq_ptrs[4] = sumsq_rows[3];
  923|  1.95k|        sum_ptrs[3] = sum_rows[2];
  924|  1.95k|        sum_ptrs[4] = sum_rows[3];
  925|       |
  926|  1.95k|        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
  927|  1.95k|        left++;
  928|  1.95k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.95k|#define PXSTRIDE(x) (x)
  ------------------
  929|       |
  930|  1.95k|        if (--h <= 0)
  ------------------
  |  Branch (930:13): [True: 268, False: 1.68k]
  ------------------
  931|    268|            goto odd;
  932|       |
  933|  1.68k|        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
  934|  1.68k|        left++;
  935|  1.68k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.68k|#define PXSTRIDE(x) (x)
  ------------------
  936|       |
  937|  1.68k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  938|  1.68k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.68k|#define BITDEPTH_MAX 0xff
  ------------------
  939|  1.68k|        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  940|  1.68k|                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  941|       |
  942|  1.68k|        if (--h <= 0)
  ------------------
  |  Branch (942:13): [True: 294, False: 1.39k]
  ------------------
  943|    294|            goto vert_2;
  944|       |
  945|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
  946|       |        // one of them to point at the previously unused rows[4].
  947|  1.39k|        sumsq_ptrs[3] = sumsq_rows[4];
  948|  1.39k|        sum_ptrs[3] = sum_rows[4];
  949|  1.39k|    }
  950|       |
  951|  66.9k|    do {
  952|  66.9k|        sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], left, src, w, edges);
  953|  66.9k|        left++;
  954|  66.9k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  66.9k|#define PXSTRIDE(x) (x)
  ------------------
  955|       |
  956|  66.9k|        if (--h <= 0)
  ------------------
  |  Branch (956:13): [True: 320, False: 66.6k]
  ------------------
  957|    320|            goto odd;
  958|       |
  959|  66.6k|        sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], left, src, w, edges);
  960|  66.6k|        left++;
  961|  66.6k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  66.6k|#define PXSTRIDE(x) (x)
  ------------------
  962|       |
  963|  66.6k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  964|  66.6k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  66.6k|#define BITDEPTH_MAX 0xff
  ------------------
  965|  66.6k|        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  966|  66.6k|                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  967|  66.6k|    } while (--h > 0);
  ------------------
  |  Branch (967:14): [True: 63.9k, False: 2.67k]
  ------------------
  968|       |
  969|  2.67k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (969:9): [True: 421, False: 2.25k]
  ------------------
  970|    421|        goto vert_2;
  971|       |
  972|  2.25k|    sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], NULL, lpf_bottom, w, edges);
  973|  2.25k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.25k|#define PXSTRIDE(x) (x)
  ------------------
  974|  2.25k|    sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], NULL, lpf_bottom, w, edges);
  975|       |
  976|  3.57k|output_2:
  977|  3.57k|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  978|  3.57k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  3.57k|#define BITDEPTH_MAX 0xff
  ------------------
  979|  3.57k|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  980|  3.57k|                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  981|  3.57k|    return;
  982|       |
  983|  1.32k|vert_2:
  984|       |    // Duplicate the last row twice more
  985|  1.32k|    sumsq_ptrs[3] = sumsq_ptrs[2];
  986|  1.32k|    sumsq_ptrs[4] = sumsq_ptrs[2];
  987|  1.32k|    sum_ptrs[3] = sum_ptrs[2];
  988|  1.32k|    sum_ptrs[4] = sum_ptrs[2];
  989|  1.32k|    goto output_2;
  990|       |
  991|    588|odd:
  992|       |    // Copy the last row as padding once
  993|    588|    sumsq_ptrs[4] = sumsq_ptrs[3];
  994|    588|    sum_ptrs[4] = sum_ptrs[3];
  995|       |
  996|    588|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  997|    588|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    588|#define BITDEPTH_MAX 0xff
  ------------------
  998|    588|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  999|    588|                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
 1000|       |
 1001|  1.22k|output_1:
 1002|       |    // Duplicate the last row twice more
 1003|  1.22k|    sumsq_ptrs[3] = sumsq_ptrs[2];
 1004|  1.22k|    sumsq_ptrs[4] = sumsq_ptrs[2];
 1005|  1.22k|    sum_ptrs[3] = sum_ptrs[2];
 1006|  1.22k|    sum_ptrs[4] = sum_ptrs[2];
 1007|       |
 1008|  1.22k|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
 1009|  1.22k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.22k|#define BITDEPTH_MAX 0xff
  ------------------
 1010|       |    // Output only one row
 1011|  1.22k|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
 1012|  1.22k|                w, 1, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
 1013|  1.22k|    return;
 1014|       |
 1015|    638|vert_1:
 1016|       |    // Copy the last row as padding once
 1017|    638|    sumsq_ptrs[4] = sumsq_ptrs[3];
 1018|    638|    sum_ptrs[4] = sum_ptrs[3];
 1019|       |
 1020|    638|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
 1021|    638|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    638|#define BITDEPTH_MAX 0xff
  ------------------
 1022|    638|    rotate(A_ptrs, B_ptrs, 2);
 1023|       |
 1024|    638|    goto output_1;
 1025|    588|}
looprestoration_tmpl.c:sgr_box5_row_h:
  441|   343k|{
  442|   343k|    sumsq++;
  443|   343k|    sum++;
  444|   343k|    int a = edges & LR_HAVE_LEFT ? (left ? left[0][1] : src[-3]) : src[0];
  ------------------
  |  Branch (444:13): [True: 249k, False: 93.7k]
  |  Branch (444:37): [True: 236k, False: 13.2k]
  ------------------
  445|   343k|    int b = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
  ------------------
  |  Branch (445:13): [True: 249k, False: 93.7k]
  |  Branch (445:37): [True: 236k, False: 13.2k]
  ------------------
  446|   343k|    int c = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
  ------------------
  |  Branch (446:13): [True: 249k, False: 93.7k]
  |  Branch (446:37): [True: 236k, False: 13.2k]
  ------------------
  447|   343k|    int d = src[0];
  448|  40.0M|    for (int x = -1; x < w + 1; x++) {
  ------------------
  |  Branch (448:22): [True: 39.7M, False: 343k]
  ------------------
  449|  39.7M|        int e = (x + 2 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 2] : src[w - 1];
  ------------------
  |  Branch (449:18): [True: 38.6M, False: 1.02M]
  |  Branch (449:31): [True: 747k, False: 281k]
  ------------------
  450|  39.7M|        sum[x] = a + b + c + d + e;
  451|  39.7M|        sumsq[x] = a * a + b * b + c * c + d * d + e * e;
  452|  39.7M|        a = b;
  453|  39.7M|        b = c;
  454|  39.7M|        c = d;
  455|  39.7M|        d = e;
  456|  39.7M|    }
  457|   343k|}
looprestoration_tmpl.c:sgr_box5_vert:
  537|   173k|{
  538|   173k|    sgr_box5_row_v(sumsq, sum, sumsq_out, sum_out, w);
  539|   173k|    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 25, 164);
  540|   173k|    rotate5_x2(sumsq, sum);
  541|   173k|}
looprestoration_tmpl.c:sgr_box5_row_v:
  488|   173k|{
  489|  20.0M|    for (int x = 0; x < w + 2; x++) {
  ------------------
  |  Branch (489:21): [True: 19.9M, False: 173k]
  ------------------
  490|  19.9M|        int sq_a = sumsq[0][x];
  491|  19.9M|        int sq_b = sumsq[1][x];
  492|  19.9M|        int sq_c = sumsq[2][x];
  493|  19.9M|        int sq_d = sumsq[3][x];
  494|  19.9M|        int sq_e = sumsq[4][x];
  495|  19.9M|        int s_a = sum[0][x];
  496|  19.9M|        int s_b = sum[1][x];
  497|  19.9M|        int s_c = sum[2][x];
  498|  19.9M|        int s_d = sum[3][x];
  499|  19.9M|        int s_e = sum[4][x];
  500|  19.9M|        sumsq_out[x] = sq_a + sq_b + sq_c + sq_d + sq_e;
  501|  19.9M|        sum_out[x] = s_a + s_b + s_c + s_d + s_e;
  502|  19.9M|    }
  503|   173k|}
looprestoration_tmpl.c:sgr_calc_row_ab:
  507|   465k|{
  508|   465k|    const int bitdepth_min_8 = bitdepth_from_max(bitdepth_max) - 8;
  ------------------
  |  |   58|   465k|#define bitdepth_from_max(x) 8
  ------------------
  509|  51.1M|    for (int i = 0; i < w + 2; i++) {
  ------------------
  |  Branch (509:21): [True: 50.6M, False: 465k]
  ------------------
  510|  50.6M|        const int a =
  511|  50.6M|            (AA[i] + ((1 << (2 * bitdepth_min_8)) >> 1)) >> (2 * bitdepth_min_8);
  512|  50.6M|        const int b =
  513|  50.6M|            (BB[i] + ((1 << bitdepth_min_8) >> 1)) >> bitdepth_min_8;
  514|       |
  515|  50.6M|        const unsigned p = imax(a * n - b * b, 0);
  516|  50.6M|        const unsigned z = (p * s + (1 << 19)) >> 20;
  517|  50.6M|        const unsigned x = dav1d_sgr_x_by_x[umin(z, 255)];
  518|       |
  519|       |        // This is where we invert A and B, so that B is of size coef.
  520|  50.6M|        AA[i] = (x * BB[i] * sgr_one_by_x + (1 << 11)) >> 12;
  521|  50.6M|        BB[i] = x;
  522|  50.6M|    }
  523|   465k|}
looprestoration_tmpl.c:rotate5_x2:
  402|   173k|{
  403|   173k|    int32_t *tmp32[2];
  404|   173k|    coef *tmpc[2];
  405|   520k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (405:21): [True: 347k, False: 173k]
  ------------------
  406|   347k|        tmp32[i] = sumsq_ptrs[i];
  407|   347k|        tmpc[i] = sum_ptrs[i];
  408|   347k|    }
  409|   694k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (409:21): [True: 520k, False: 173k]
  ------------------
  410|   520k|        sumsq_ptrs[i] = sumsq_ptrs[i + 2];
  411|   520k|        sum_ptrs[i] = sum_ptrs[i + 2];
  412|   520k|    }
  413|   520k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (413:21): [True: 347k, False: 173k]
  ------------------
  414|   347k|        sumsq_ptrs[3 + i] = tmp32[i];
  415|   347k|        sum_ptrs[3 + i] = tmpc[i];
  416|   347k|    }
  417|   173k|}
looprestoration_tmpl.c:rotate:
  390|   757k|{
  391|   757k|    int32_t *tmp32 = sumsq_ptrs[0];
  392|   757k|    coef *tmpc = sum_ptrs[0];
  393|  2.29M|    for (int i = 0; i < n - 1; i++) {
  ------------------
  |  Branch (393:21): [True: 1.53M, False: 757k]
  ------------------
  394|  1.53M|        sumsq_ptrs[i] = sumsq_ptrs[i + 1];
  395|  1.53M|        sum_ptrs[i] = sum_ptrs[i + 1];
  396|  1.53M|    }
  397|   757k|    sumsq_ptrs[n - 1] = tmp32;
  398|   757k|    sum_ptrs[n - 1] = tmpc;
  399|   757k|}
looprestoration_tmpl.c:sgr_finish2:
  645|  73.6k|{
  646|  73.6k|    ALIGN_STK_16(coef, tmp, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|  73.6k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  73.6k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  647|       |
  648|  73.6k|    sgr_finish_filter2(tmp, *dst, stride, A_ptrs, B_ptrs, w, h);
  649|  73.6k|    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
  650|  73.6k|    *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|  73.6k|#define PXSTRIDE(x) (x)
  ------------------
  651|  73.6k|    if (h > 1) {
  ------------------
  |  Branch (651:9): [True: 72.4k, False: 1.22k]
  ------------------
  652|  72.4k|        sgr_weighted_row1(*dst, tmp + FILTER_OUT_STRIDE, w, w1 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |  572|  72.4k|#define FILTER_OUT_STRIDE (384)
  ------------------
  653|  72.4k|        *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|  72.4k|#define PXSTRIDE(x) (x)
  ------------------
  654|  72.4k|    }
  655|  73.6k|    rotate(A_ptrs, B_ptrs, 2);
  656|  73.6k|}
looprestoration_tmpl.c:sgr_finish_filter2:
  579|   162k|{
  580|   162k|#define SIX_NEIGHBORS(P, i)\
  581|   162k|    ((P[0][i]     + P[1][i]) * 6 +   \
  582|   162k|     (P[0][i - 1] + P[1][i - 1] +    \
  583|   162k|      P[0][i + 1] + P[1][i + 1]) * 5)
  584|  18.6M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (584:21): [True: 18.5M, False: 162k]
  ------------------
  585|  18.5M|        const int a = SIX_NEIGHBORS(B_ptrs, i + 1);
  ------------------
  |  |  581|  18.5M|    ((P[0][i]     + P[1][i]) * 6 +   \
  |  |  582|  18.5M|     (P[0][i - 1] + P[1][i - 1] +    \
  |  |  583|  18.5M|      P[0][i + 1] + P[1][i + 1]) * 5)
  ------------------
  586|  18.5M|        const int b = SIX_NEIGHBORS(A_ptrs, i + 1);
  ------------------
  |  |  581|  18.5M|    ((P[0][i]     + P[1][i]) * 6 +   \
  |  |  582|  18.5M|     (P[0][i - 1] + P[1][i - 1] +    \
  |  |  583|  18.5M|      P[0][i + 1] + P[1][i + 1]) * 5)
  ------------------
  587|  18.5M|        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
  588|  18.5M|    }
  589|   162k|    if (h <= 1)
  ------------------
  |  Branch (589:9): [True: 2.59k, False: 160k]
  ------------------
  590|  2.59k|        return;
  591|   160k|    tmp += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|   160k|#define FILTER_OUT_STRIDE (384)
  ------------------
  592|   160k|    src += PXSTRIDE(src_stride);
  ------------------
  |  |   53|   160k|#define PXSTRIDE(x) (x)
  ------------------
  593|   160k|    const int32_t *A = &A_ptrs[1][1];
  594|   160k|    const coef *B = &B_ptrs[1][1];
  595|  18.4M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (595:21): [True: 18.2M, False: 160k]
  ------------------
  596|  18.2M|        const int a = B[i] * 6 + (B[i - 1] + B[i + 1]) * 5;
  597|  18.2M|        const int b = A[i] * 6 + (A[i - 1] + A[i + 1]) * 5;
  598|  18.2M|        tmp[i] = (b - a * src[i] + (1 << 7)) >> 8;
  599|  18.2M|    }
  600|   160k|#undef SIX_NEIGHBORS
  601|   160k|}
looprestoration_tmpl.c:sgr_weighted_row1:
  605|   242k|{
  606|  29.4M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (606:21): [True: 29.2M, False: 242k]
  ------------------
  607|  29.2M|        const int v = w1 * t1[i];
  608|  29.2M|        dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
  ------------------
  |  |   49|  29.2M|#define iclip_pixel iclip_u8
  ------------------
  609|  29.2M|    }
  610|   242k|}
looprestoration_tmpl.c:sgr_3x3_c:
  684|  3.24k|{
  685|  3.24k|#define BUF_STRIDE (384 + 16)
  686|  3.24k|    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  3.24k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  3.24k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  687|  3.24k|    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  3.24k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  3.24k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  688|  3.24k|    int32_t *sumsq_ptrs[3], *sumsq_rows[3];
  689|  3.24k|    coef *sum_ptrs[3], *sum_rows[3];
  690|  12.9k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (690:21): [True: 9.72k, False: 3.24k]
  ------------------
  691|  9.72k|        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.72k|#define BUF_STRIDE (384 + 16)
  ------------------
  692|  9.72k|        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.72k|#define BUF_STRIDE (384 + 16)
  ------------------
  693|  9.72k|    }
  694|       |
  695|  3.24k|    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  3.24k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  3.24k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  696|  3.24k|    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  3.24k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  3.24k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  697|  3.24k|    int32_t *A_ptrs[3];
  698|  3.24k|    coef *B_ptrs[3];
  699|  12.9k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (699:21): [True: 9.72k, False: 3.24k]
  ------------------
  700|  9.72k|        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.72k|#define BUF_STRIDE (384 + 16)
  ------------------
  701|  9.72k|        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  9.72k|#define BUF_STRIDE (384 + 16)
  ------------------
  702|  9.72k|    }
  703|  3.24k|    const pixel *src = dst;
  704|  3.24k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  3.24k|#define PXSTRIDE(x) (x)
  ------------------
  705|       |
  706|  3.24k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (706:9): [True: 1.54k, False: 1.69k]
  ------------------
  707|  1.54k|        sumsq_ptrs[0] = sumsq_rows[0];
  708|  1.54k|        sumsq_ptrs[1] = sumsq_rows[1];
  709|  1.54k|        sumsq_ptrs[2] = sumsq_rows[2];
  710|  1.54k|        sum_ptrs[0] = sum_rows[0];
  711|  1.54k|        sum_ptrs[1] = sum_rows[1];
  712|  1.54k|        sum_ptrs[2] = sum_rows[2];
  713|       |
  714|  1.54k|        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
  715|  1.54k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.54k|#define PXSTRIDE(x) (x)
  ------------------
  716|  1.54k|        sgr_box3_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
  717|       |
  718|  1.54k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  719|  1.54k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.54k|#define BITDEPTH_MAX 0xff
  ------------------
  720|  1.54k|        left++;
  721|  1.54k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.54k|#define PXSTRIDE(x) (x)
  ------------------
  722|  1.54k|        rotate(A_ptrs, B_ptrs, 3);
  723|       |
  724|  1.54k|        if (--h <= 0)
  ------------------
  |  Branch (724:13): [True: 379, False: 1.16k]
  ------------------
  725|    379|            goto vert_1;
  726|       |
  727|  1.16k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  728|  1.16k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.16k|#define BITDEPTH_MAX 0xff
  ------------------
  729|  1.16k|        left++;
  730|  1.16k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.16k|#define PXSTRIDE(x) (x)
  ------------------
  731|  1.16k|        rotate(A_ptrs, B_ptrs, 3);
  732|       |
  733|  1.16k|        if (--h <= 0)
  ------------------
  |  Branch (733:13): [True: 136, False: 1.03k]
  ------------------
  734|    136|            goto vert_2;
  735|  1.69k|    } else {
  736|  1.69k|        sumsq_ptrs[0] = sumsq_rows[0];
  737|  1.69k|        sumsq_ptrs[1] = sumsq_rows[0];
  738|  1.69k|        sumsq_ptrs[2] = sumsq_rows[0];
  739|  1.69k|        sum_ptrs[0] = sum_rows[0];
  740|  1.69k|        sum_ptrs[1] = sum_rows[0];
  741|  1.69k|        sum_ptrs[2] = sum_rows[0];
  742|       |
  743|  1.69k|        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
  744|  1.69k|        left++;
  745|  1.69k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.69k|#define PXSTRIDE(x) (x)
  ------------------
  746|       |
  747|  1.69k|        sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  748|  1.69k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.69k|#define BITDEPTH_MAX 0xff
  ------------------
  749|  1.69k|        rotate(A_ptrs, B_ptrs, 3);
  750|       |
  751|  1.69k|        if (--h <= 0)
  ------------------
  |  Branch (751:13): [True: 287, False: 1.40k]
  ------------------
  752|    287|            goto vert_1;
  753|       |
  754|  1.40k|        sumsq_ptrs[2] = sumsq_rows[1];
  755|  1.40k|        sum_ptrs[2] = sum_rows[1];
  756|       |
  757|  1.40k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  758|  1.40k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.40k|#define BITDEPTH_MAX 0xff
  ------------------
  759|  1.40k|        left++;
  760|  1.40k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.40k|#define PXSTRIDE(x) (x)
  ------------------
  761|  1.40k|        rotate(A_ptrs, B_ptrs, 3);
  762|       |
  763|  1.40k|        if (--h <= 0)
  ------------------
  |  Branch (763:13): [True: 219, False: 1.18k]
  ------------------
  764|    219|            goto vert_2;
  765|       |
  766|  1.18k|        sumsq_ptrs[2] = sumsq_rows[2];
  767|  1.18k|        sum_ptrs[2] = sum_rows[2];
  768|  1.18k|    }
  769|       |
  770|  90.5k|    do {
  771|  90.5k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  772|  90.5k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  90.5k|#define BITDEPTH_MAX 0xff
  ------------------
  773|  90.5k|        left++;
  774|  90.5k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  90.5k|#define PXSTRIDE(x) (x)
  ------------------
  775|       |
  776|  90.5k|        sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  777|  90.5k|                    w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  778|  90.5k|    } while (--h > 0);
  ------------------
  |  Branch (778:14): [True: 88.3k, False: 2.21k]
  ------------------
  779|       |
  780|  2.21k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (780:9): [True: 658, False: 1.56k]
  ------------------
  781|    658|        goto vert_2;
  782|       |
  783|  1.56k|    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  784|  1.56k|                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.56k|#define BITDEPTH_MAX 0xff
  ------------------
  785|  1.56k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  1.56k|#define PXSTRIDE(x) (x)
  ------------------
  786|       |
  787|  1.56k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  788|  1.56k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  789|       |
  790|  1.56k|    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  791|  1.56k|                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.56k|#define BITDEPTH_MAX 0xff
  ------------------
  792|       |
  793|  1.56k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  794|  1.56k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  795|  1.56k|    return;
  796|       |
  797|  1.01k|vert_2:
  798|  1.01k|    sumsq_ptrs[2] = sumsq_ptrs[1];
  799|  1.01k|    sum_ptrs[2] = sum_ptrs[1];
  800|  1.01k|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  801|  1.01k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.01k|#define BITDEPTH_MAX 0xff
  ------------------
  802|       |
  803|  1.01k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  804|  1.01k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  805|       |
  806|  1.67k|output_1:
  807|  1.67k|    sumsq_ptrs[2] = sumsq_ptrs[1];
  808|  1.67k|    sum_ptrs[2] = sum_ptrs[1];
  809|  1.67k|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  810|  1.67k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.67k|#define BITDEPTH_MAX 0xff
  ------------------
  811|       |
  812|  1.67k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  813|  1.67k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  814|  1.67k|    return;
  815|       |
  816|    666|vert_1:
  817|    666|    sumsq_ptrs[2] = sumsq_ptrs[1];
  818|    666|    sum_ptrs[2] = sum_ptrs[1];
  819|    666|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  820|    666|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|    666|#define BITDEPTH_MAX 0xff
  ------------------
  821|    666|    rotate(A_ptrs, B_ptrs, 3);
  822|    666|    goto output_1;
  823|  1.01k|}
looprestoration_tmpl.c:sgr_box3_row_h:
  423|   290k|{
  424|   290k|    sumsq++;
  425|   290k|    sum++;
  426|   290k|    int a = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
  ------------------
  |  Branch (426:13): [True: 209k, False: 80.7k]
  |  Branch (426:37): [True: 198k, False: 11.2k]
  ------------------
  427|   290k|    int b = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
  ------------------
  |  Branch (427:13): [True: 209k, False: 80.7k]
  |  Branch (427:37): [True: 198k, False: 11.2k]
  ------------------
  428|  31.1M|    for (int x = -1; x < w + 1; x++) {
  ------------------
  |  Branch (428:22): [True: 30.8M, False: 290k]
  ------------------
  429|  30.8M|        int c = (x + 1 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 1] : src[w - 1];
  ------------------
  |  Branch (429:18): [True: 30.2M, False: 581k]
  |  Branch (429:31): [True: 402k, False: 178k]
  ------------------
  430|  30.8M|        sum[x] = a + b + c;
  431|  30.8M|        sumsq[x] = a * a + b * b + c * c;
  432|  30.8M|        a = b;
  433|  30.8M|        b = c;
  434|  30.8M|    }
  435|   290k|}
looprestoration_tmpl.c:sgr_box3_hv:
  550|  97.7k|{
  551|  97.7k|    sgr_box3_row_h(sumsq[2], sum[2], left, src, w, edges);
  552|  97.7k|    sgr_box3_vert(sumsq, sum, AA, BB, w, s, bitdepth_max);
  553|  97.7k|}
looprestoration_tmpl.c:sgr_box3_vert:
  528|   291k|{
  529|   291k|    sgr_box3_row_v(sumsq, sum, sumsq_out, sum_out, w);
  530|   291k|    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 9, 455);
  531|   291k|    rotate(sumsq, sum, 3);
  532|   291k|}
looprestoration_tmpl.c:sgr_box3_row_v:
  472|   291k|{
  473|  31.0M|    for (int x = 0; x < w + 2; x++) {
  ------------------
  |  Branch (473:21): [True: 30.7M, False: 291k]
  ------------------
  474|  30.7M|        int sq_a = sumsq[0][x];
  475|  30.7M|        int sq_b = sumsq[1][x];
  476|  30.7M|        int sq_c = sumsq[2][x];
  477|  30.7M|        int s_a = sum[0][x];
  478|  30.7M|        int s_b = sum[1][x];
  479|  30.7M|        int s_c = sum[2][x];
  480|  30.7M|        sumsq_out[x] = sq_a + sq_b + sq_c;
  481|  30.7M|        sum_out[x] = s_a + s_b + s_c;
  482|  30.7M|    }
  483|   291k|}
looprestoration_tmpl.c:sgr_finish1:
  631|  96.3k|{
  632|       |    // Only one single row, no stride needed
  633|  96.3k|    ALIGN_STK_16(coef, tmp, 384,);
  ------------------
  |  |  100|  96.3k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  96.3k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  634|       |
  635|  96.3k|    sgr_finish_filter_row1(tmp, *dst, A_ptrs, B_ptrs, w);
  636|  96.3k|    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
  637|  96.3k|    *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|  96.3k|#define PXSTRIDE(x) (x)
  ------------------
  638|  96.3k|    rotate(A_ptrs, B_ptrs, 3);
  639|  96.3k|}
looprestoration_tmpl.c:sgr_finish_filter_row1:
  559|   272k|{
  560|   272k|#define EIGHT_NEIGHBORS(P, i)\
  561|   272k|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  562|   272k|     (P[0][i - 1] + P[2][i - 1] +                           \
  563|   272k|      P[0][i + 1] + P[2][i + 1]) * 3)
  564|  28.7M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (564:21): [True: 28.5M, False: 272k]
  ------------------
  565|  28.5M|        const int a = EIGHT_NEIGHBORS(B_ptrs, i + 1);
  ------------------
  |  |  561|  28.5M|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  |  |  562|  28.5M|     (P[0][i - 1] + P[2][i - 1] +                           \
  |  |  563|  28.5M|      P[0][i + 1] + P[2][i + 1]) * 3)
  ------------------
  566|  28.5M|        const int b = EIGHT_NEIGHBORS(A_ptrs, i + 1);
  ------------------
  |  |  561|  28.5M|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  |  |  562|  28.5M|     (P[0][i - 1] + P[2][i - 1] +                           \
  |  |  563|  28.5M|      P[0][i + 1] + P[2][i + 1]) * 3)
  ------------------
  567|  28.5M|        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
  568|  28.5M|    }
  569|   272k|#undef EIGHT_NEIGHBORS
  570|   272k|}
looprestoration_tmpl.c:sgr_mix_c:
 1032|  6.02k|{
 1033|  6.02k|    ALIGN_STK_16(int32_t, sumsq5_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1034|  6.02k|    ALIGN_STK_16(coef, sum5_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1035|  6.02k|    int32_t *sumsq5_ptrs[5], *sumsq5_rows[5];
 1036|  6.02k|    coef *sum5_ptrs[5], *sum5_rows[5];
 1037|  36.1k|    for (int i = 0; i < 5; i++) {
  ------------------
  |  Branch (1037:21): [True: 30.1k, False: 6.02k]
  ------------------
 1038|  30.1k|        sumsq5_rows[i] = &sumsq5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  30.1k|#define BUF_STRIDE (384 + 16)
  ------------------
 1039|  30.1k|        sum5_rows[i] = &sum5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  30.1k|#define BUF_STRIDE (384 + 16)
  ------------------
 1040|  30.1k|    }
 1041|  6.02k|    ALIGN_STK_16(int32_t, sumsq3_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1042|  6.02k|    ALIGN_STK_16(coef, sum3_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1043|  6.02k|    int32_t *sumsq3_ptrs[3], *sumsq3_rows[3];
 1044|  6.02k|    coef *sum3_ptrs[3], *sum3_rows[3];
 1045|  24.0k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (1045:21): [True: 18.0k, False: 6.02k]
  ------------------
 1046|  18.0k|        sumsq3_rows[i] = &sumsq3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  18.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1047|  18.0k|        sum3_rows[i] = &sum3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  18.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1048|  18.0k|    }
 1049|       |
 1050|  6.02k|    ALIGN_STK_16(int32_t, A5_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1051|  6.02k|    ALIGN_STK_16(coef, B5_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1052|  6.02k|    int32_t *A5_ptrs[2];
 1053|  6.02k|    coef *B5_ptrs[2];
 1054|  18.0k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1054:21): [True: 12.0k, False: 6.02k]
  ------------------
 1055|  12.0k|        A5_ptrs[i] = &A5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  12.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1056|  12.0k|        B5_ptrs[i] = &B5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  12.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1057|  12.0k|    }
 1058|  6.02k|    ALIGN_STK_16(int32_t, A3_buf, BUF_STRIDE * 4 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1059|  6.02k|    ALIGN_STK_16(coef, B3_buf, BUF_STRIDE * 4 + 16,);
  ------------------
  |  |  100|  6.02k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  6.02k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1060|  6.02k|    int32_t *A3_ptrs[4];
 1061|  6.02k|    coef *B3_ptrs[4];
 1062|  30.1k|    for (int i = 0; i < 4; i++) {
  ------------------
  |  Branch (1062:21): [True: 24.0k, False: 6.02k]
  ------------------
 1063|  24.0k|        A3_ptrs[i] = &A3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  24.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1064|  24.0k|        B3_ptrs[i] = &B3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  24.0k|#define BUF_STRIDE (384 + 16)
  ------------------
 1065|  24.0k|    }
 1066|  6.02k|    const pixel *src = dst;
 1067|  6.02k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  6.02k|#define PXSTRIDE(x) (x)
  ------------------
 1068|       |
 1069|  6.02k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (1069:9): [True: 2.84k, False: 3.17k]
  ------------------
 1070|  2.84k|        sumsq5_ptrs[0] = sumsq5_rows[0];
 1071|  2.84k|        sumsq5_ptrs[1] = sumsq5_rows[0];
 1072|  2.84k|        sumsq5_ptrs[2] = sumsq5_rows[1];
 1073|  2.84k|        sumsq5_ptrs[3] = sumsq5_rows[2];
 1074|  2.84k|        sumsq5_ptrs[4] = sumsq5_rows[3];
 1075|  2.84k|        sum5_ptrs[0] = sum5_rows[0];
 1076|  2.84k|        sum5_ptrs[1] = sum5_rows[0];
 1077|  2.84k|        sum5_ptrs[2] = sum5_rows[1];
 1078|  2.84k|        sum5_ptrs[3] = sum5_rows[2];
 1079|  2.84k|        sum5_ptrs[4] = sum5_rows[3];
 1080|       |
 1081|  2.84k|        sumsq3_ptrs[0] = sumsq3_rows[0];
 1082|  2.84k|        sumsq3_ptrs[1] = sumsq3_rows[1];
 1083|  2.84k|        sumsq3_ptrs[2] = sumsq3_rows[2];
 1084|  2.84k|        sum3_ptrs[0] = sum3_rows[0];
 1085|  2.84k|        sum3_ptrs[1] = sum3_rows[1];
 1086|  2.84k|        sum3_ptrs[2] = sum3_rows[2];
 1087|       |
 1088|  2.84k|        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
 1089|  2.84k|                        sumsq5_rows[0], sum5_rows[0],
 1090|  2.84k|                        NULL, lpf, w, edges);
 1091|  2.84k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.84k|#define PXSTRIDE(x) (x)
  ------------------
 1092|  2.84k|        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
 1093|  2.84k|                        sumsq5_rows[1], sum5_rows[1],
 1094|  2.84k|                        NULL, lpf, w, edges);
 1095|       |
 1096|  2.84k|        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
 1097|  2.84k|                        sumsq5_rows[2], sum5_rows[2],
 1098|  2.84k|                        left, src, w, edges);
 1099|  2.84k|        left++;
 1100|  2.84k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.84k|#define PXSTRIDE(x) (x)
  ------------------
 1101|       |
 1102|  2.84k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1103|  2.84k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.84k|#define BITDEPTH_MAX 0xff
  ------------------
 1104|  2.84k|        rotate(A3_ptrs, B3_ptrs, 4);
 1105|       |
 1106|  2.84k|        if (--h <= 0)
  ------------------
  |  Branch (1106:13): [True: 479, False: 2.36k]
  ------------------
 1107|    479|            goto vert_1;
 1108|       |
 1109|  2.36k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1110|  2.36k|                        sumsq5_rows[3], sum5_rows[3],
 1111|  2.36k|                        left, src, w, edges);
 1112|  2.36k|        left++;
 1113|  2.36k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.36k|#define PXSTRIDE(x) (x)
  ------------------
 1114|  2.36k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1115|  2.36k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.36k|#define BITDEPTH_MAX 0xff
  ------------------
 1116|  2.36k|        rotate(A5_ptrs, B5_ptrs, 2);
 1117|  2.36k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1118|  2.36k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.36k|#define BITDEPTH_MAX 0xff
  ------------------
 1119|  2.36k|        rotate(A3_ptrs, B3_ptrs, 4);
 1120|       |
 1121|  2.36k|        if (--h <= 0)
  ------------------
  |  Branch (1121:13): [True: 333, False: 2.03k]
  ------------------
 1122|    333|            goto vert_2;
 1123|       |
 1124|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
 1125|       |        // one of them to point at the previously unused rows[4].
 1126|  2.03k|        sumsq5_ptrs[3] = sumsq5_rows[4];
 1127|  2.03k|        sum5_ptrs[3] = sum5_rows[4];
 1128|  3.17k|    } else {
 1129|  3.17k|        sumsq5_ptrs[0] = sumsq5_rows[0];
 1130|  3.17k|        sumsq5_ptrs[1] = sumsq5_rows[0];
 1131|  3.17k|        sumsq5_ptrs[2] = sumsq5_rows[0];
 1132|  3.17k|        sumsq5_ptrs[3] = sumsq5_rows[0];
 1133|  3.17k|        sumsq5_ptrs[4] = sumsq5_rows[0];
 1134|  3.17k|        sum5_ptrs[0] = sum5_rows[0];
 1135|  3.17k|        sum5_ptrs[1] = sum5_rows[0];
 1136|  3.17k|        sum5_ptrs[2] = sum5_rows[0];
 1137|  3.17k|        sum5_ptrs[3] = sum5_rows[0];
 1138|  3.17k|        sum5_ptrs[4] = sum5_rows[0];
 1139|       |
 1140|  3.17k|        sumsq3_ptrs[0] = sumsq3_rows[0];
 1141|  3.17k|        sumsq3_ptrs[1] = sumsq3_rows[0];
 1142|  3.17k|        sumsq3_ptrs[2] = sumsq3_rows[0];
 1143|  3.17k|        sum3_ptrs[0] = sum3_rows[0];
 1144|  3.17k|        sum3_ptrs[1] = sum3_rows[0];
 1145|  3.17k|        sum3_ptrs[2] = sum3_rows[0];
 1146|       |
 1147|  3.17k|        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
 1148|  3.17k|                        sumsq5_rows[0], sum5_rows[0],
 1149|  3.17k|                        left, src, w, edges);
 1150|  3.17k|        left++;
 1151|  3.17k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  3.17k|#define PXSTRIDE(x) (x)
  ------------------
 1152|       |
 1153|  3.17k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1154|  3.17k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  3.17k|#define BITDEPTH_MAX 0xff
  ------------------
 1155|  3.17k|        rotate(A3_ptrs, B3_ptrs, 4);
 1156|       |
 1157|  3.17k|        if (--h <= 0)
  ------------------
  |  Branch (1157:13): [True: 355, False: 2.82k]
  ------------------
 1158|    355|            goto vert_1;
 1159|       |
 1160|  2.82k|        sumsq5_ptrs[4] = sumsq5_rows[1];
 1161|  2.82k|        sum5_ptrs[4] = sum5_rows[1];
 1162|       |
 1163|  2.82k|        sumsq3_ptrs[2] = sumsq3_rows[1];
 1164|  2.82k|        sum3_ptrs[2] = sum3_rows[1];
 1165|       |
 1166|  2.82k|        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
 1167|  2.82k|                        sumsq5_rows[1], sum5_rows[1],
 1168|  2.82k|                        left, src, w, edges);
 1169|  2.82k|        left++;
 1170|  2.82k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.82k|#define PXSTRIDE(x) (x)
  ------------------
 1171|       |
 1172|  2.82k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1173|  2.82k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.82k|#define BITDEPTH_MAX 0xff
  ------------------
 1174|  2.82k|        rotate(A5_ptrs, B5_ptrs, 2);
 1175|  2.82k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1176|  2.82k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.82k|#define BITDEPTH_MAX 0xff
  ------------------
 1177|  2.82k|        rotate(A3_ptrs, B3_ptrs, 4);
 1178|       |
 1179|  2.82k|        if (--h <= 0)
  ------------------
  |  Branch (1179:13): [True: 429, False: 2.39k]
  ------------------
 1180|    429|            goto vert_2;
 1181|       |
 1182|  2.39k|        sumsq5_ptrs[3] = sumsq5_rows[2];
 1183|  2.39k|        sumsq5_ptrs[4] = sumsq5_rows[3];
 1184|  2.39k|        sum5_ptrs[3] = sum5_rows[2];
 1185|  2.39k|        sum5_ptrs[4] = sum5_rows[3];
 1186|       |
 1187|  2.39k|        sumsq3_ptrs[2] = sumsq3_rows[2];
 1188|  2.39k|        sum3_ptrs[2] = sum3_rows[2];
 1189|       |
 1190|  2.39k|        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
 1191|  2.39k|                        sumsq5_rows[2], sum5_rows[2],
 1192|  2.39k|                        left, src, w, edges);
 1193|  2.39k|        left++;
 1194|  2.39k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.39k|#define PXSTRIDE(x) (x)
  ------------------
 1195|       |
 1196|  2.39k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1197|  2.39k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.39k|#define BITDEPTH_MAX 0xff
  ------------------
 1198|  2.39k|        rotate(A3_ptrs, B3_ptrs, 4);
 1199|       |
 1200|  2.39k|        if (--h <= 0)
  ------------------
  |  Branch (1200:13): [True: 259, False: 2.13k]
  ------------------
 1201|    259|            goto odd;
 1202|       |
 1203|  2.13k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1204|  2.13k|                        sumsq5_rows[3], sum5_rows[3],
 1205|  2.13k|                        left, src, w, edges);
 1206|  2.13k|        left++;
 1207|  2.13k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.13k|#define PXSTRIDE(x) (x)
  ------------------
 1208|       |
 1209|  2.13k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1210|  2.13k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.13k|#define BITDEPTH_MAX 0xff
  ------------------
 1211|  2.13k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1212|  2.13k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.13k|#define BITDEPTH_MAX 0xff
  ------------------
 1213|  2.13k|        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1214|  2.13k|                       w, 2, params->sgr.w0, params->sgr.w1
 1215|  2.13k|                       HIGHBD_TAIL_SUFFIX);
 1216|       |
 1217|  2.13k|        if (--h <= 0)
  ------------------
  |  Branch (1217:13): [True: 304, False: 1.83k]
  ------------------
 1218|    304|            goto vert_2;
 1219|       |
 1220|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
 1221|       |        // one of them to point at the previously unused rows[4].
 1222|  1.83k|        sumsq5_ptrs[3] = sumsq5_rows[4];
 1223|  1.83k|        sum5_ptrs[3] = sum5_rows[4];
 1224|  1.83k|    }
 1225|       |
 1226|  80.5k|    do {
 1227|  80.5k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1228|  80.5k|                        sumsq5_ptrs[3], sum5_ptrs[3],
 1229|  80.5k|                        left, src, w, edges);
 1230|  80.5k|        left++;
 1231|  80.5k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  80.5k|#define PXSTRIDE(x) (x)
  ------------------
 1232|       |
 1233|  80.5k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1234|  80.5k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  80.5k|#define BITDEPTH_MAX 0xff
  ------------------
 1235|  80.5k|        rotate(A3_ptrs, B3_ptrs, 4);
 1236|       |
 1237|  80.5k|        if (--h <= 0)
  ------------------
  |  Branch (1237:13): [True: 277, False: 80.3k]
  ------------------
 1238|    277|            goto odd;
 1239|       |
 1240|  80.3k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1241|  80.3k|                        sumsq5_ptrs[4], sum5_ptrs[4],
 1242|  80.3k|                        left, src, w, edges);
 1243|  80.3k|        left++;
 1244|  80.3k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  80.3k|#define PXSTRIDE(x) (x)
  ------------------
 1245|       |
 1246|  80.3k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1247|  80.3k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  80.3k|#define BITDEPTH_MAX 0xff
  ------------------
 1248|  80.3k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1249|  80.3k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  80.3k|#define BITDEPTH_MAX 0xff
  ------------------
 1250|  80.3k|        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1251|  80.3k|                       w, 2, params->sgr.w0, params->sgr.w1
 1252|  80.3k|                       HIGHBD_TAIL_SUFFIX);
 1253|  80.3k|    } while (--h > 0);
  ------------------
  |  Branch (1253:14): [True: 76.7k, False: 3.58k]
  ------------------
 1254|       |
 1255|  3.58k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (1255:9): [True: 731, False: 2.85k]
  ------------------
 1256|    731|        goto vert_2;
 1257|       |
 1258|  2.85k|    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1259|  2.85k|                    sumsq5_ptrs[3], sum5_ptrs[3],
 1260|  2.85k|                    NULL, lpf_bottom, w, edges);
 1261|  2.85k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.85k|#define PXSTRIDE(x) (x)
  ------------------
 1262|  2.85k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1263|  2.85k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.85k|#define BITDEPTH_MAX 0xff
  ------------------
 1264|  2.85k|    rotate(A3_ptrs, B3_ptrs, 4);
 1265|       |
 1266|  2.85k|    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1267|  2.85k|                    sumsq5_ptrs[4], sum5_ptrs[4],
 1268|  2.85k|                    NULL, lpf_bottom, w, edges);
 1269|       |
 1270|  4.65k|output_2:
 1271|  4.65k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1272|  4.65k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  4.65k|#define BITDEPTH_MAX 0xff
  ------------------
 1273|  4.65k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1274|  4.65k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  4.65k|#define BITDEPTH_MAX 0xff
  ------------------
 1275|  4.65k|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1276|  4.65k|                   w, 2, params->sgr.w0, params->sgr.w1
 1277|  4.65k|                   HIGHBD_TAIL_SUFFIX);
 1278|  4.65k|    return;
 1279|       |
 1280|  1.79k|vert_2:
 1281|       |    // Duplicate the last row twice more
 1282|  1.79k|    sumsq5_ptrs[3] = sumsq5_ptrs[2];
 1283|  1.79k|    sumsq5_ptrs[4] = sumsq5_ptrs[2];
 1284|  1.79k|    sum5_ptrs[3] = sum5_ptrs[2];
 1285|  1.79k|    sum5_ptrs[4] = sum5_ptrs[2];
 1286|       |
 1287|  1.79k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1288|  1.79k|    sum3_ptrs[2] = sum3_ptrs[1];
 1289|  1.79k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1290|  1.79k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.79k|#define BITDEPTH_MAX 0xff
  ------------------
 1291|  1.79k|    rotate(A3_ptrs, B3_ptrs, 4);
 1292|       |
 1293|  1.79k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1294|  1.79k|    sum3_ptrs[2] = sum3_ptrs[1];
 1295|       |
 1296|  1.79k|    goto output_2;
 1297|       |
 1298|    536|odd:
 1299|       |    // Copy the last row as padding once
 1300|    536|    sumsq5_ptrs[4] = sumsq5_ptrs[3];
 1301|    536|    sum5_ptrs[4] = sum5_ptrs[3];
 1302|       |
 1303|    536|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1304|    536|    sum3_ptrs[2] = sum3_ptrs[1];
 1305|       |
 1306|    536|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1307|    536|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    536|#define BITDEPTH_MAX 0xff
  ------------------
 1308|    536|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1309|    536|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|    536|#define BITDEPTH_MAX 0xff
  ------------------
 1310|    536|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1311|    536|                   w, 2, params->sgr.w0, params->sgr.w1
 1312|    536|                   HIGHBD_TAIL_SUFFIX);
 1313|       |
 1314|  1.37k|output_1:
 1315|       |    // Duplicate the last row twice more
 1316|  1.37k|    sumsq5_ptrs[3] = sumsq5_ptrs[2];
 1317|  1.37k|    sumsq5_ptrs[4] = sumsq5_ptrs[2];
 1318|  1.37k|    sum5_ptrs[3] = sum5_ptrs[2];
 1319|  1.37k|    sum5_ptrs[4] = sum5_ptrs[2];
 1320|       |
 1321|  1.37k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1322|  1.37k|    sum3_ptrs[2] = sum3_ptrs[1];
 1323|       |
 1324|  1.37k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1325|  1.37k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.37k|#define BITDEPTH_MAX 0xff
  ------------------
 1326|  1.37k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1327|  1.37k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.37k|#define BITDEPTH_MAX 0xff
  ------------------
 1328|  1.37k|    rotate(A3_ptrs, B3_ptrs, 4);
 1329|       |    // Output only one row
 1330|  1.37k|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1331|  1.37k|                   w, 1, params->sgr.w0, params->sgr.w1
 1332|  1.37k|                   HIGHBD_TAIL_SUFFIX);
 1333|  1.37k|    return;
 1334|       |
 1335|    834|vert_1:
 1336|       |    // Copy the last row as padding once
 1337|    834|    sumsq5_ptrs[4] = sumsq5_ptrs[3];
 1338|    834|    sum5_ptrs[4] = sum5_ptrs[3];
 1339|       |
 1340|    834|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1341|    834|    sum3_ptrs[2] = sum3_ptrs[1];
 1342|       |
 1343|    834|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1344|    834|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    834|#define BITDEPTH_MAX 0xff
  ------------------
 1345|    834|    rotate(A5_ptrs, B5_ptrs, 2);
 1346|    834|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1347|    834|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|    834|#define BITDEPTH_MAX 0xff
  ------------------
 1348|    834|    rotate(A3_ptrs, B3_ptrs, 4);
 1349|       |
 1350|    834|    goto output_1;
 1351|    536|}
looprestoration_tmpl.c:sgr_box35_row_h:
  464|   188k|{
  465|   188k|    sgr_box3_row_h(sumsq3, sum3, left, src, w, edges);
  466|   188k|    sgr_box5_row_h(sumsq5, sum5, left, src, w, edges);
  467|   188k|}
looprestoration_tmpl.c:sgr_finish_mix:
  663|  88.9k|{
  664|  88.9k|    ALIGN_STK_16(coef, tmp5, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|  88.9k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  88.9k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  665|  88.9k|    ALIGN_STK_16(coef, tmp3, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|  88.9k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  88.9k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  666|       |
  667|  88.9k|    sgr_finish_filter2(tmp5, *dst, stride, A5_ptrs, B5_ptrs, w, h);
  668|  88.9k|    sgr_finish_filter_row1(tmp3, *dst, A3_ptrs, B3_ptrs, w);
  669|  88.9k|    if (h > 1)
  ------------------
  |  Branch (669:9): [True: 87.6k, False: 1.37k]
  ------------------
  670|  87.6k|        sgr_finish_filter_row1(tmp3 + FILTER_OUT_STRIDE, *dst + PXSTRIDE(stride),
  ------------------
  |  |  572|  87.6k|#define FILTER_OUT_STRIDE (384)
  ------------------
                      sgr_finish_filter_row1(tmp3 + FILTER_OUT_STRIDE, *dst + PXSTRIDE(stride),
  ------------------
  |  |   53|  87.6k|#define PXSTRIDE(x) (x)
  ------------------
  671|  87.6k|                               &A3_ptrs[1], &B3_ptrs[1], w);
  672|  88.9k|    sgr_weighted2(*dst, stride, tmp5, tmp3, w, h, w0, w1 HIGHBD_TAIL_SUFFIX);
  673|  88.9k|    *dst += h*PXSTRIDE(stride);
  ------------------
  |  |   53|  88.9k|#define PXSTRIDE(x) (x)
  ------------------
  674|  88.9k|    rotate(A5_ptrs, B5_ptrs, 2);
  675|  88.9k|    rotate(A3_ptrs, B3_ptrs, 4);
  676|  88.9k|}
looprestoration_tmpl.c:sgr_weighted2:
  616|  88.9k|{
  617|   265k|    for (int j = 0; j < h; j++) {
  ------------------
  |  Branch (617:21): [True: 176k, False: 88.9k]
  ------------------
  618|  18.2M|        for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (618:25): [True: 18.0M, False: 176k]
  ------------------
  619|  18.0M|            const int v = w0 * t1[i] + w1 * t2[i];
  620|  18.0M|            dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
  ------------------
  |  |   49|  18.0M|#define iclip_pixel iclip_u8
  ------------------
  621|  18.0M|        }
  622|   176k|        dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|   176k|#define PXSTRIDE(x) (x)
  ------------------
  623|   176k|        t1 += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|   176k|#define FILTER_OUT_STRIDE (384)
  ------------------
  624|   176k|        t2 += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|   176k|#define FILTER_OUT_STRIDE (384)
  ------------------
  625|   176k|    }
  626|  88.9k|}
dav1d_loop_restoration_dsp_init_16bpc:
 1367|  4.99k|{
 1368|  4.99k|    c->wiener[0] = c->wiener[1] = wiener_c;
 1369|  4.99k|    c->sgr[0] = sgr_5x5_c;
 1370|  4.99k|    c->sgr[1] = sgr_3x3_c;
 1371|  4.99k|    c->sgr[2] = sgr_mix_c;
 1372|       |
 1373|  4.99k|#if HAVE_ASM
 1374|       |#if ARCH_AARCH64 || ARCH_ARM
 1375|       |    loop_restoration_dsp_init_arm(c, bpc);
 1376|       |#elif ARCH_LOONGARCH64
 1377|       |    loop_restoration_dsp_init_loongarch(c, bpc);
 1378|       |#elif ARCH_PPC64LE
 1379|       |    loop_restoration_dsp_init_ppc(c, bpc);
 1380|       |#elif ARCH_X86
 1381|       |    loop_restoration_dsp_init_x86(c, bpc);
 1382|  4.99k|#endif
 1383|  4.99k|#endif
 1384|  4.99k|}

dav1d_lr_sbrow_8bpc:
  171|  21.3k|{
  172|  21.3k|    const int offset_y = 8 * !!sby;
  173|  21.3k|    const ptrdiff_t *const dst_stride = f->sr_cur.p.stride;
  174|  21.3k|    const int restore_planes = f->lf.restore_planes;
  175|  21.3k|    const int sb128 = f->seq_hdr->sb128;
  176|  21.3k|    const int not_last = sby + 1 < f->sbh;
  177|       |
  178|  21.3k|    if (restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (178:9): [True: 16.7k, False: 4.61k]
  ------------------
  179|  16.7k|        const int h = f->sr_cur.p.p.h;
  180|  16.7k|        const int w = f->sr_cur.p.p.w;
  181|  16.7k|        const int next_row_y = (sby + 1) << (6 + sb128);
  182|  16.7k|        const int row_h = imin(next_row_y - 8 * not_last, h);
  183|  16.7k|        const int y_stripe = (sby << (6 + sb128)) - offset_y;
  184|  16.7k|        lr_sbrow(f, dst[0] - offset_y * PXSTRIDE(dst_stride[0]), y_stripe, w,
  ------------------
  |  |   53|  16.7k|#define PXSTRIDE(x) (x)
  ------------------
  185|  16.7k|                 h, row_h, 0);
  186|  16.7k|    }
  187|  21.3k|    if (restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) {
  ------------------
  |  Branch (187:9): [True: 7.21k, False: 14.1k]
  ------------------
  188|  7.21k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  189|  7.21k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  190|  7.21k|        const int h = (f->sr_cur.p.p.h + ss_ver) >> ss_ver;
  191|  7.21k|        const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
  192|  7.21k|        const int next_row_y = (sby + 1) << ((6 - ss_ver) + sb128);
  193|  7.21k|        const int row_h = imin(next_row_y - (8 >> ss_ver) * not_last, h);
  194|  7.21k|        const int offset_uv = offset_y >> ss_ver;
  195|  7.21k|        const int y_stripe = (sby << ((6 - ss_ver) + sb128)) - offset_uv;
  196|  7.21k|        if (restore_planes & LR_RESTORE_U)
  ------------------
  |  Branch (196:13): [True: 2.78k, False: 4.43k]
  ------------------
  197|  2.78k|            lr_sbrow(f, dst[1] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  ------------------
  |  |   53|  2.78k|#define PXSTRIDE(x) (x)
  ------------------
  198|  2.78k|                     w, h, row_h, 1);
  199|       |
  200|  7.21k|        if (restore_planes & LR_RESTORE_V)
  ------------------
  |  Branch (200:13): [True: 5.86k, False: 1.35k]
  ------------------
  201|  5.86k|            lr_sbrow(f, dst[2] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  ------------------
  |  |   53|  5.86k|#define PXSTRIDE(x) (x)
  ------------------
  202|  5.86k|                     w, h, row_h, 2);
  203|  7.21k|    }
  204|  21.3k|}
lr_apply_tmpl.c:lr_sbrow:
  109|  47.7k|{
  110|  47.7k|    const int chroma = !!plane;
  111|  47.7k|    const int ss_ver = chroma & (f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420);
  112|  47.7k|    const int ss_hor = chroma & (f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444);
  113|  47.7k|    const ptrdiff_t p_stride = f->sr_cur.p.stride[chroma];
  114|       |
  115|  47.7k|    const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!plane];
  116|  47.7k|    const int unit_size = 1 << unit_size_log2;
  117|  47.7k|    const int half_unit_size = unit_size >> 1;
  118|  47.7k|    const int max_unit_size = unit_size + half_unit_size;
  119|       |
  120|       |    // Y coordinate of the sbrow (y is 8 luma pixel rows above row_y)
  121|  47.7k|    const int row_y = y + ((8 >> ss_ver) * !!y);
  122|       |
  123|       |    // FIXME This is an ugly hack to lookup the proper AV1Filter unit for
  124|       |    // chroma planes. Question: For Multithreaded decoding, is it better
  125|       |    // to store the chroma LR information with collocated Luma information?
  126|       |    // In other words. For a chroma restoration unit locate at 128,128 and
  127|       |    // with a 4:2:0 chroma subsampling, do we store the filter information at
  128|       |    // the AV1Filter unit located at (128,128) or (256,256)
  129|       |    // TODO Support chroma subsampling.
  130|  47.7k|    const int shift_hor = 7 - ss_hor;
  131|       |
  132|       |    /* maximum sbrow height is 128 + 8 rows offset */
  133|  47.7k|    ALIGN_STK_16(pixel, pre_lr_border, 2, [128 + 8][4]);
  ------------------
  |  |  100|  47.7k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  47.7k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  134|  47.7k|    const Av1RestorationUnit *lr[2];
  135|       |
  136|  47.7k|    enum LrEdgeFlags edges = (y > 0 ? LR_HAVE_TOP : 0) | LR_HAVE_RIGHT;
  ------------------
  |  Branch (136:31): [True: 32.4k, False: 15.2k]
  ------------------
  137|       |
  138|  47.7k|    int aligned_unit_pos = row_y & ~(unit_size - 1);
  139|  47.7k|    if (aligned_unit_pos && aligned_unit_pos + half_unit_size > h)
  ------------------
  |  Branch (139:9): [True: 30.3k, False: 17.3k]
  |  Branch (139:29): [True: 738, False: 29.6k]
  ------------------
  140|    738|        aligned_unit_pos -= unit_size;
  141|  47.7k|    aligned_unit_pos <<= ss_ver;
  142|  47.7k|    const int sb_idx = (aligned_unit_pos >> 7) * f->sr_sb128w;
  143|  47.7k|    const int unit_idx = ((aligned_unit_pos >> 6) & 1) << 1;
  144|  47.7k|    lr[0] = &f->lf.lr_mask[sb_idx].lr[plane][unit_idx];
  145|  47.7k|    int restore = lr[0]->type != DAV1D_RESTORATION_NONE;
  146|  47.7k|    int x = 0, bit = 0;
  147|  47.7k|    const int backup_h = row_h - y;
  148|  70.4k|    for (; x + max_unit_size <= w; p += unit_size, edges |= LR_HAVE_LEFT, bit ^= 1) {
  ------------------
  |  Branch (148:12): [True: 22.7k, False: 47.7k]
  ------------------
  149|  22.7k|        const int next_x = x + unit_size;
  150|  22.7k|        const int next_u_idx = unit_idx + ((next_x >> (shift_hor - 1)) & 1);
  151|  22.7k|        lr[!bit] =
  152|  22.7k|            &f->lf.lr_mask[sb_idx + (next_x >> shift_hor)].lr[plane][next_u_idx];
  153|  22.7k|        const int restore_next = lr[!bit]->type != DAV1D_RESTORATION_NONE;
  154|  22.7k|        if (restore_next)
  ------------------
  |  Branch (154:13): [True: 11.6k, False: 11.1k]
  ------------------
  155|  11.6k|            backup4xU(pre_lr_border[bit], p + unit_size - 4, p_stride, backup_h);
  156|  22.7k|        if (restore)
  ------------------
  |  Branch (156:13): [True: 11.5k, False: 11.2k]
  ------------------
  157|  11.5k|            lr_stripe(f, p, pre_lr_border[!bit], x, y, plane, unit_size, row_h,
  158|  11.5k|                      lr[bit], edges);
  159|  22.7k|        x = next_x;
  160|  22.7k|        restore = restore_next;
  161|  22.7k|    }
  162|  47.7k|    if (restore) {
  ------------------
  |  Branch (162:9): [True: 8.40k, False: 39.3k]
  ------------------
  163|  8.40k|        edges &= ~LR_HAVE_RIGHT;
  164|  8.40k|        const int unit_w = w - x;
  165|  8.40k|        lr_stripe(f, p, pre_lr_border[!bit], x, y, plane, unit_w, row_h, lr[bit], edges);
  166|  8.40k|    }
  167|  47.7k|}
lr_apply_tmpl.c:backup4xU:
  102|  11.6k|{
  103|   671k|    for (; u > 0; u--, dst++, src += PXSTRIDE(src_stride))
  ------------------
  |  |   53|   659k|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (103:12): [True: 659k, False: 11.6k]
  ------------------
  104|   659k|        pixel_copy(dst, src, 4);
  ------------------
  |  |   47|   659k|#define pixel_copy memcpy
  ------------------
  105|  11.6k|}
lr_apply_tmpl.c:lr_stripe:
   40|  19.9k|{
   41|  19.9k|    const Dav1dDSPContext *const dsp = f->dsp;
   42|  19.9k|    const int chroma = !!plane;
   43|  19.9k|    const int ss_ver = chroma & (f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420);
   44|  19.9k|    const ptrdiff_t stride = f->sr_cur.p.stride[chroma];
   45|  19.9k|    const int sby = (y + (y ? 8 << ss_ver : 0)) >> (6 - ss_ver + f->seq_hdr->sb128);
  ------------------
  |  Branch (45:27): [True: 9.44k, False: 10.5k]
  ------------------
   46|  19.9k|    const int have_tt = f->c->n_tc > 1;
   47|  19.9k|    const pixel *lpf = f->lf.lr_lpf_line[plane] +
   48|  19.9k|        have_tt * (sby * (4 << f->seq_hdr->sb128) - 4) * PXSTRIDE(stride) + x;
  ------------------
  |  |   53|  19.9k|#define PXSTRIDE(x) (x)
  ------------------
   49|       |
   50|       |    // The first stripe of the frame is shorter by 8 luma pixel rows.
   51|  19.9k|    int stripe_h = imin((64 - 8 * !y) >> ss_ver, row_h - y);
   52|       |
   53|  19.9k|    looprestorationfilter_fn lr_fn;
   54|  19.9k|    LooprestorationParams params;
   55|  19.9k|    if (lr->type == DAV1D_RESTORATION_WIENER) {
  ------------------
  |  Branch (55:9): [True: 5.00k, False: 14.9k]
  ------------------
   56|  5.00k|        int16_t (*const filter)[8] = params.filter;
   57|  5.00k|        filter[0][0] = filter[0][6] = lr->filter_h[0];
   58|  5.00k|        filter[0][1] = filter[0][5] = lr->filter_h[1];
   59|  5.00k|        filter[0][2] = filter[0][4] = lr->filter_h[2];
   60|  5.00k|        filter[0][3] = -(filter[0][0] + filter[0][1] + filter[0][2]) * 2;
   61|       |#if BITDEPTH != 8
   62|       |        /* For 8-bit SIMD it's beneficial to handle the +128 separately
   63|       |         * in order to avoid overflows. */
   64|       |        filter[0][3] += 128;
   65|       |#endif
   66|       |
   67|  5.00k|        filter[1][0] = filter[1][6] = lr->filter_v[0];
   68|  5.00k|        filter[1][1] = filter[1][5] = lr->filter_v[1];
   69|  5.00k|        filter[1][2] = filter[1][4] = lr->filter_v[2];
   70|  5.00k|        filter[1][3] = 128 - (filter[1][0] + filter[1][1] + filter[1][2]) * 2;
   71|       |
   72|  5.00k|        lr_fn = dsp->lr.wiener[!(filter[0][0] | filter[1][0])];
   73|  14.9k|    } else {
   74|  14.9k|        assert(lr->type >= DAV1D_RESTORATION_SGRPROJ);
  ------------------
  |  Branch (74:9): [True: 14.9k, False: 0]
  ------------------
   75|  14.9k|        const int sgr_idx = lr->type - DAV1D_RESTORATION_SGRPROJ;
   76|  14.9k|        const uint16_t *const sgr_params = dav1d_sgr_params[sgr_idx];
   77|  14.9k|        params.sgr.s0 = sgr_params[0];
   78|  14.9k|        params.sgr.s1 = sgr_params[1];
   79|  14.9k|        params.sgr.w0 = lr->sgr_weights[0];
   80|  14.9k|        params.sgr.w1 = 128 - (lr->sgr_weights[0] + lr->sgr_weights[1]);
   81|       |
   82|  14.9k|        lr_fn = dsp->lr.sgr[!!sgr_params[0] + !!sgr_params[1] * 2 - 1];
   83|  14.9k|    }
   84|       |
   85|  32.2k|    while (y + stripe_h <= row_h) {
  ------------------
  |  Branch (85:12): [True: 32.2k, False: 0]
  ------------------
   86|       |        // Change the HAVE_BOTTOM bit in edges to (sby + 1 != f->sbh || y + stripe_h != row_h)
   87|  32.2k|        edges ^= (-(sby + 1 != f->sbh || y + stripe_h != row_h) ^ edges) & LR_HAVE_BOTTOM;
  ------------------
  |  Branch (87:21): [True: 16.2k, False: 16.0k]
  |  Branch (87:42): [True: 5.62k, False: 10.3k]
  ------------------
   88|  32.2k|        lr_fn(p, stride, left, lpf, unit_w, stripe_h, &params, edges HIGHBD_CALL_SUFFIX);
   89|       |
   90|  32.2k|        left += stripe_h;
   91|  32.2k|        y += stripe_h;
   92|  32.2k|        p += stripe_h * PXSTRIDE(stride);
  ------------------
  |  |   53|  32.2k|#define PXSTRIDE(x) (x)
  ------------------
   93|  32.2k|        edges |= LR_HAVE_TOP;
   94|  32.2k|        stripe_h = imin(64 >> ss_ver, row_h - y);
   95|  32.2k|        if (stripe_h == 0) break;
  ------------------
  |  Branch (95:13): [True: 19.9k, False: 12.2k]
  ------------------
   96|  12.2k|        lpf += 4 * PXSTRIDE(stride);
  ------------------
  |  |   53|  12.2k|#define PXSTRIDE(x) (x)
  ------------------
   97|  12.2k|    }
   98|  19.9k|}
dav1d_lr_sbrow_16bpc:
  171|  15.1k|{
  172|  15.1k|    const int offset_y = 8 * !!sby;
  173|  15.1k|    const ptrdiff_t *const dst_stride = f->sr_cur.p.stride;
  174|  15.1k|    const int restore_planes = f->lf.restore_planes;
  175|  15.1k|    const int sb128 = f->seq_hdr->sb128;
  176|  15.1k|    const int not_last = sby + 1 < f->sbh;
  177|       |
  178|  15.1k|    if (restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (178:9): [True: 12.4k, False: 2.60k]
  ------------------
  179|  12.4k|        const int h = f->sr_cur.p.p.h;
  180|  12.4k|        const int w = f->sr_cur.p.p.w;
  181|  12.4k|        const int next_row_y = (sby + 1) << (6 + sb128);
  182|  12.4k|        const int row_h = imin(next_row_y - 8 * not_last, h);
  183|  12.4k|        const int y_stripe = (sby << (6 + sb128)) - offset_y;
  184|  12.4k|        lr_sbrow(f, dst[0] - offset_y * PXSTRIDE(dst_stride[0]), y_stripe, w,
  185|  12.4k|                 h, row_h, 0);
  186|  12.4k|    }
  187|  15.1k|    if (restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) {
  ------------------
  |  Branch (187:9): [True: 6.91k, False: 8.18k]
  ------------------
  188|  6.91k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  189|  6.91k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  190|  6.91k|        const int h = (f->sr_cur.p.p.h + ss_ver) >> ss_ver;
  191|  6.91k|        const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
  192|  6.91k|        const int next_row_y = (sby + 1) << ((6 - ss_ver) + sb128);
  193|  6.91k|        const int row_h = imin(next_row_y - (8 >> ss_ver) * not_last, h);
  194|  6.91k|        const int offset_uv = offset_y >> ss_ver;
  195|  6.91k|        const int y_stripe = (sby << ((6 - ss_ver) + sb128)) - offset_uv;
  196|  6.91k|        if (restore_planes & LR_RESTORE_U)
  ------------------
  |  Branch (196:13): [True: 5.30k, False: 1.61k]
  ------------------
  197|  5.30k|            lr_sbrow(f, dst[1] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  198|  5.30k|                     w, h, row_h, 1);
  199|       |
  200|  6.91k|        if (restore_planes & LR_RESTORE_V)
  ------------------
  |  Branch (200:13): [True: 4.56k, False: 2.35k]
  ------------------
  201|  4.56k|            lr_sbrow(f, dst[2] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  202|  4.56k|                     w, h, row_h, 2);
  203|  6.91k|    }
  204|  15.1k|}

dav1d_mc_dsp_init_8bpc:
  960|  3.66k|COLD void bitfn(dav1d_mc_dsp_init)(Dav1dMCDSPContext *const c) {
  961|  3.66k|#define init_mc_fns(type, name) do { \
  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  966|  3.66k|} while (0)
  967|       |
  968|  3.66k|    init_mc_fns(FILTER_2D_8TAP_REGULAR,        8tap_regular);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  969|  3.66k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  970|  3.66k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  971|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  972|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  973|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SHARP,          8tap_sharp);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  974|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  975|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH,         8tap_smooth);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  976|  3.66k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  977|  3.66k|    init_mc_fns(FILTER_2D_BILINEAR,            bilin);
  ------------------
  |  |  961|  3.66k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.66k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.66k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.66k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.66k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.66k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.66k]
  |  |  ------------------
  ------------------
  978|       |
  979|  3.66k|    c->avg      = avg_c;
  980|  3.66k|    c->w_avg    = w_avg_c;
  981|  3.66k|    c->mask     = mask_c;
  982|  3.66k|    c->blend    = blend_c;
  983|  3.66k|    c->blend_v  = blend_v_c;
  984|  3.66k|    c->blend_h  = blend_h_c;
  985|  3.66k|    c->w_mask[0] = w_mask_444_c;
  986|  3.66k|    c->w_mask[1] = w_mask_422_c;
  987|  3.66k|    c->w_mask[2] = w_mask_420_c;
  988|  3.66k|    c->warp8x8  = warp_affine_8x8_c;
  989|  3.66k|    c->warp8x8t = warp_affine_8x8t_c;
  990|  3.66k|    c->emu_edge = emu_edge_c;
  991|  3.66k|    c->resize   = resize_c;
  992|       |
  993|  3.66k|#if HAVE_ASM
  994|       |#if ARCH_AARCH64 || ARCH_ARM
  995|       |    mc_dsp_init_arm(c);
  996|       |#elif ARCH_LOONGARCH64
  997|       |    mc_dsp_init_loongarch(c);
  998|       |#elif ARCH_PPC64LE
  999|       |    mc_dsp_init_ppc(c);
 1000|       |#elif ARCH_RISCV
 1001|       |    mc_dsp_init_riscv(c);
 1002|       |#elif ARCH_X86
 1003|       |    mc_dsp_init_x86(c);
 1004|  3.66k|#endif
 1005|  3.66k|#endif
 1006|  3.66k|}
dav1d_mc_dsp_init_16bpc:
  960|  4.99k|COLD void bitfn(dav1d_mc_dsp_init)(Dav1dMCDSPContext *const c) {
  961|  4.99k|#define init_mc_fns(type, name) do { \
  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  966|  4.99k|} while (0)
  967|       |
  968|  4.99k|    init_mc_fns(FILTER_2D_8TAP_REGULAR,        8tap_regular);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  969|  4.99k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  970|  4.99k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  971|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  972|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  973|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SHARP,          8tap_sharp);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  974|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  975|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH,         8tap_smooth);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  976|  4.99k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  977|  4.99k|    init_mc_fns(FILTER_2D_BILINEAR,            bilin);
  ------------------
  |  |  961|  4.99k|#define init_mc_fns(type, name) do { \
  |  |  962|  4.99k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  4.99k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  4.99k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  4.99k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  4.99k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 4.99k]
  |  |  ------------------
  ------------------
  978|       |
  979|  4.99k|    c->avg      = avg_c;
  980|  4.99k|    c->w_avg    = w_avg_c;
  981|  4.99k|    c->mask     = mask_c;
  982|  4.99k|    c->blend    = blend_c;
  983|  4.99k|    c->blend_v  = blend_v_c;
  984|  4.99k|    c->blend_h  = blend_h_c;
  985|  4.99k|    c->w_mask[0] = w_mask_444_c;
  986|  4.99k|    c->w_mask[1] = w_mask_422_c;
  987|  4.99k|    c->w_mask[2] = w_mask_420_c;
  988|  4.99k|    c->warp8x8  = warp_affine_8x8_c;
  989|  4.99k|    c->warp8x8t = warp_affine_8x8t_c;
  990|  4.99k|    c->emu_edge = emu_edge_c;
  991|  4.99k|    c->resize   = resize_c;
  992|       |
  993|  4.99k|#if HAVE_ASM
  994|       |#if ARCH_AARCH64 || ARCH_ARM
  995|       |    mc_dsp_init_arm(c);
  996|       |#elif ARCH_LOONGARCH64
  997|       |    mc_dsp_init_loongarch(c);
  998|       |#elif ARCH_PPC64LE
  999|       |    mc_dsp_init_ppc(c);
 1000|       |#elif ARCH_RISCV
 1001|       |    mc_dsp_init_riscv(c);
 1002|       |#elif ARCH_X86
 1003|       |    mc_dsp_init_x86(c);
 1004|  4.99k|#endif
 1005|  4.99k|#endif
 1006|  4.99k|}

dav1d_mem_pool_push:
  224|   235k|void dav1d_mem_pool_push(Dav1dMemPool *const pool, void *const ptr) {
  225|   235k|    pthread_mutex_lock(&pool->lock);
  226|   235k|    Dav1dMemPoolBuffer *const buf = (Dav1dMemPoolBuffer*)((uintptr_t)ptr - 64);
  227|   235k|    const int ref_cnt = --pool->ref_cnt;
  228|   235k|    if (!pool->end) {
  ------------------
  |  Branch (228:9): [True: 234k, False: 1.00k]
  ------------------
  229|   234k|        buf->next = pool->buf;
  230|   234k|        pool->buf = buf;
  231|   234k|        pthread_mutex_unlock(&pool->lock);
  232|   234k|        assert(ref_cnt > 0);
  ------------------
  |  Branch (232:9): [True: 234k, False: 0]
  ------------------
  233|   234k|    } else {
  234|  1.00k|        pthread_mutex_unlock(&pool->lock);
  235|  1.00k|        dav1d_free_aligned(buf);
  ------------------
  |  |  136|  1.00k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  236|  1.00k|        if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (236:13): [True: 1.00k, False: 0]
  ------------------
  237|  1.00k|    }
  238|   235k|}
dav1d_mem_pool_pop:
  240|   235k|void *dav1d_mem_pool_pop(Dav1dMemPool *const pool, const size_t size) {
  241|   235k|    pthread_mutex_lock(&pool->lock);
  242|   235k|    Dav1dMemPoolBuffer *buf = pool->buf;
  243|   235k|    pool->ref_cnt++;
  244|       |
  245|   235k|    if (buf) {
  ------------------
  |  Branch (245:9): [True: 171k, False: 64.6k]
  ------------------
  246|   171k|        pool->buf = buf->next;
  247|   171k|        pthread_mutex_unlock(&pool->lock);
  248|   171k|        if (buf->size != size) {
  ------------------
  |  Branch (248:13): [True: 4.48k, False: 166k]
  ------------------
  249|       |            /* Reallocate if the size has changed */
  250|  4.48k|            dav1d_free_aligned(buf);
  ------------------
  |  |  136|  4.48k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  251|  4.48k|            goto alloc;
  252|  4.48k|        }
  253|       |#if TRACK_HEAP_ALLOCATIONS
  254|       |        dav1d_track_reuse(pool->type);
  255|       |#endif
  256|   171k|    } else {
  257|  64.6k|        pthread_mutex_unlock(&pool->lock);
  258|  69.0k|alloc:
  259|  69.0k|        buf = dav1d_alloc_aligned(pool->type, size + 64, 64);
  ------------------
  |  |  134|  69.0k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  260|  69.0k|        if (!buf) {
  ------------------
  |  Branch (260:13): [True: 0, False: 69.0k]
  ------------------
  261|      0|            pthread_mutex_lock(&pool->lock);
  262|      0|            const int ref_cnt = --pool->ref_cnt;
  263|      0|            pthread_mutex_unlock(&pool->lock);
  264|      0|            if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (264:17): [True: 0, False: 0]
  ------------------
  265|      0|            return NULL;
  266|      0|        }
  267|  69.0k|        buf->size = size;
  268|  69.0k|    }
  269|       |
  270|   235k|    return (void*)((uintptr_t)buf + 64);
  271|   235k|}
dav1d_mem_pool_init:
  275|  71.5k|{
  276|  71.5k|    Dav1dMemPool *const pool = dav1d_malloc(ALLOC_COMMON_CTX,
  ------------------
  |  |  132|  71.5k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
  277|  71.5k|                                            sizeof(Dav1dMemPool));
  278|  71.5k|    if (pool) {
  ------------------
  |  Branch (278:9): [True: 71.5k, False: 0]
  ------------------
  279|  71.5k|        if (!pthread_mutex_init(&pool->lock, NULL)) {
  ------------------
  |  Branch (279:13): [True: 71.5k, False: 0]
  ------------------
  280|  71.5k|            pool->buf = NULL;
  281|  71.5k|            pool->ref_cnt = 1;
  282|  71.5k|            pool->end = 0;
  283|       |#if TRACK_HEAP_ALLOCATIONS
  284|       |            pool->type = type;
  285|       |#endif
  286|  71.5k|            *ppool = pool;
  287|  71.5k|            return 0;
  288|  71.5k|        }
  289|      0|        dav1d_free(pool);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  290|      0|    }
  291|      0|    *ppool = NULL;
  292|      0|    return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  293|  71.5k|}
dav1d_mem_pool_end:
  295|  71.5k|COLD void dav1d_mem_pool_end(Dav1dMemPool *const pool) {
  296|  71.5k|    if (pool) {
  ------------------
  |  Branch (296:9): [True: 71.5k, False: 0]
  ------------------
  297|  71.5k|        pthread_mutex_lock(&pool->lock);
  298|  71.5k|        Dav1dMemPoolBuffer *buf = pool->buf;
  299|  71.5k|        const int ref_cnt = --pool->ref_cnt;
  300|  71.5k|        pool->buf = NULL;
  301|  71.5k|        pool->end = 1;
  302|  71.5k|        pthread_mutex_unlock(&pool->lock);
  303|       |
  304|   135k|        while (buf) {
  ------------------
  |  Branch (304:16): [True: 63.6k, False: 71.5k]
  ------------------
  305|  63.6k|            void *const ptr = buf;
  306|  63.6k|            buf = buf->next;
  307|  63.6k|            dav1d_free_aligned(ptr);
  ------------------
  |  |  136|  63.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  308|  63.6k|        }
  309|  71.5k|        if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (309:13): [True: 70.5k, False: 1.00k]
  ------------------
  310|  71.5k|    }
  311|  71.5k|}
mem.c:mem_pool_destroy:
  219|  71.5k|static COLD void mem_pool_destroy(Dav1dMemPool *const pool) {
  220|  71.5k|    pthread_mutex_destroy(&pool->lock);
  221|  71.5k|    dav1d_free(pool);
  ------------------
  |  |  135|  71.5k|#define dav1d_free(ptr) free(ptr)
  ------------------
  222|  71.5k|}

lib.c:dav1d_alloc_aligned_internal:
   89|  30.6k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  30.6k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 30.6k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  30.6k|    void *ptr;
   95|  30.6k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 30.6k]
  ------------------
   96|  30.6k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  30.6k|}
lib.c:dav1d_free_aligned_internal:
  113|  81.8k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  81.8k|}
lib.c:dav1d_freep_aligned:
  144|  10.2k|static inline void dav1d_freep_aligned(void *ptr) {
  145|  10.2k|    void **mem = (void **) ptr;
  146|  10.2k|    if (*mem) {
  ------------------
  |  Branch (146:9): [True: 10.2k, False: 0]
  ------------------
  147|  10.2k|        dav1d_free_aligned(*mem);
  ------------------
  |  |  136|  10.2k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  148|       |        *mem = NULL;
  149|  10.2k|    }
  150|  10.2k|}
mem.c:dav1d_free_aligned_internal:
  113|  69.0k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  69.0k|}
mem.c:dav1d_alloc_aligned_internal:
   89|  69.0k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  69.0k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 69.0k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  69.0k|    void *ptr;
   95|  69.0k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 69.0k]
  ------------------
   96|  69.0k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  69.0k|}
ref.c:dav1d_alloc_aligned_internal:
   89|  80.0k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  80.0k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 80.0k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  80.0k|    void *ptr;
   95|  80.0k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 80.0k]
  ------------------
   96|  80.0k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  80.0k|}
ref.c:dav1d_free_aligned_internal:
  113|  80.0k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  80.0k|}
refmvs.c:dav1d_free_aligned_internal:
  113|  7.15k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  7.15k|}
refmvs.c:dav1d_alloc_aligned_internal:
   89|  7.15k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  7.15k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 7.15k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  7.15k|    void *ptr;
   95|  7.15k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 7.15k]
  ------------------
   96|  7.15k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  7.15k|}
decode.c:dav1d_free_aligned_internal:
  113|  42.8k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  42.8k|}
decode.c:dav1d_alloc_aligned_internal:
   89|  42.8k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  42.8k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 42.8k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  42.8k|    void *ptr;
   95|  42.8k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 42.8k]
  ------------------
   96|  42.8k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  42.8k|}

dav1d_msac_decode_subexp:
   62|   133k|{
   63|   133k|    assert(n >> k == 8);
  ------------------
  |  Branch (63:5): [True: 133k, False: 0]
  ------------------
   64|       |
   65|   133k|    unsigned a = 0;
   66|   133k|    if (dav1d_msac_decode_bool_equi(s)) {
  ------------------
  |  |   53|   133k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (66:9): [True: 74.4k, False: 59.4k]
  ------------------
   67|  74.4k|        if (dav1d_msac_decode_bool_equi(s))
  ------------------
  |  |   53|  74.4k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (67:13): [True: 44.8k, False: 29.6k]
  ------------------
   68|  44.8k|            k += dav1d_msac_decode_bool_equi(s) + 1;
  ------------------
  |  |   53|  44.8k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   69|  74.4k|        a = 1 << k;
   70|  74.4k|    }
   71|   133k|    const unsigned v = dav1d_msac_decode_bools(s, k) + a;
   72|   133k|    return ref * 2 <= n ? inv_recenter(ref, v) :
  ------------------
  |  Branch (72:12): [True: 73.2k, False: 60.6k]
  ------------------
   73|   133k|                          n - 1 - inv_recenter(n - 1 - ref, v);
   74|   133k|}
dav1d_msac_init:
  206|  50.5k|{
  207|  50.5k|    s->buf_pos = data;
  208|  50.5k|    s->buf_end = data + sz;
  209|  50.5k|    s->dif = 0;
  210|  50.5k|    s->rng = 0x8000;
  211|  50.5k|    s->cnt = -15;
  212|  50.5k|    s->allow_update_cdf = !disable_cdf_update_flag;
  213|  50.5k|    ctx_refill(s);
  214|       |
  215|  50.5k|#if ARCH_X86_64 && HAVE_ASM
  216|  50.5k|    s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt_c;
  217|       |
  218|  50.5k|    msac_init_x86(s);
  219|  50.5k|#endif
  220|  50.5k|}
msac.c:ctx_refill:
   41|  50.5k|static inline void ctx_refill(MsacContext *const s) {
   42|  50.5k|    const uint8_t *buf_pos = s->buf_pos;
   43|  50.5k|    const uint8_t *buf_end = s->buf_end;
   44|  50.5k|    int c = EC_WIN_SIZE - s->cnt - 24;
  ------------------
  |  |   39|  50.5k|#define EC_WIN_SIZE (sizeof(ec_win) << 3)
  ------------------
   45|  50.5k|    ec_win dif = s->dif;
   46|   170k|    do {
   47|   170k|        if (buf_pos >= buf_end) {
  ------------------
  |  Branch (47:13): [True: 43.8k, False: 126k]
  ------------------
   48|       |            // set remaining bits to 1;
   49|  43.8k|            dif |= ~(~(ec_win)0xff << c);
   50|  43.8k|            break;
   51|  43.8k|        }
   52|   126k|        dif |= (ec_win)(*buf_pos++ ^ 0xff) << c;
   53|   126k|        c -= 8;
   54|   126k|    } while (c >= 0);
  ------------------
  |  Branch (54:14): [True: 119k, False: 6.68k]
  ------------------
   55|  50.5k|    s->dif = dif;
   56|  50.5k|    s->cnt = EC_WIN_SIZE - c - 24;
  ------------------
  |  |   39|  50.5k|#define EC_WIN_SIZE (sizeof(ec_win) << 3)
  ------------------
   57|  50.5k|    s->buf_pos = buf_pos;
   58|  50.5k|}

decode.c:dav1d_msac_decode_bools:
   94|   565k|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|   565k|    unsigned v = 0;
   96|  1.30M|    while (n--)
  ------------------
  |  Branch (96:12): [True: 736k, False: 565k]
  ------------------
   97|   736k|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|   736k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|   565k|    return v;
   99|   565k|}
decode.c:dav1d_msac_decode_uniform:
  101|  96.7k|static inline int dav1d_msac_decode_uniform(MsacContext *const s, const unsigned n) {
  102|  96.7k|    assert(n > 0);
  ------------------
  |  Branch (102:5): [True: 96.7k, False: 0]
  ------------------
  103|  96.7k|    const int l = ulog2(n) + 1;
  104|  96.7k|    assert(l > 1);
  ------------------
  |  Branch (104:5): [True: 96.7k, False: 0]
  ------------------
  105|  96.7k|    const unsigned m = (1 << l) - n;
  106|  96.7k|    const unsigned v = dav1d_msac_decode_bools(s, l - 1);
  107|  96.7k|    return v < m ? v : (v << 1) - m + dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|  26.3k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (107:12): [True: 70.3k, False: 26.3k]
  ------------------
  108|  96.7k|}
msac.c:dav1d_msac_decode_bools:
   94|   133k|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|   133k|    unsigned v = 0;
   96|   605k|    while (n--)
  ------------------
  |  Branch (96:12): [True: 471k, False: 133k]
  ------------------
   97|   471k|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|   471k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|   133k|    return v;
   99|   133k|}
recon_tmpl.c:dav1d_msac_decode_bools:
   94|  3.21M|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|  3.21M|    unsigned v = 0;
   96|  12.4M|    while (n--)
  ------------------
  |  Branch (96:12): [True: 9.24M, False: 3.21M]
  ------------------
   97|  9.24M|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|  9.24M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|  3.21M|    return v;
   99|  3.21M|}

dav1d_parse_sequence_header:
  304|  13.1k|{
  305|  13.1k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  13.1k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 13.1k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  306|  13.1k|    validate_input_or_ret(ptr != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  13.1k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 13.1k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  307|  13.1k|    validate_input_or_ret(sz > 0 && sz <= SIZE_MAX / 2, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  26.3k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 13.1k, False: 0]
  |  |  |  Branch (52:11): [True: 13.1k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  308|       |
  309|  13.1k|    GetBits gb;
  310|  13.1k|    dav1d_init_get_bits(&gb, ptr, sz);
  311|  13.1k|    int res = DAV1D_ERR(ENOENT);
  ------------------
  |  |   58|  13.1k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  312|       |
  313|  28.7k|    do {
  314|  28.7k|        dav1d_get_bit(&gb); // obu_forbidden_bit
  315|  28.7k|        const enum Dav1dObuType type = dav1d_get_bits(&gb, 4);
  316|  28.7k|        const int has_extension = dav1d_get_bit(&gb);
  317|  28.7k|        const int has_length_field = dav1d_get_bit(&gb);
  318|  28.7k|        dav1d_get_bits(&gb, 1 + 8 * has_extension); // ignore
  319|       |
  320|  28.7k|        const uint8_t *obu_end = gb.ptr_end;
  321|  28.7k|        if (has_length_field) {
  ------------------
  |  Branch (321:13): [True: 17.5k, False: 11.2k]
  ------------------
  322|  17.5k|            const size_t len = dav1d_get_uleb128(&gb);
  323|  17.5k|            if (len > (size_t)(obu_end - gb.ptr)) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    251|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (323:17): [True: 251, False: 17.2k]
  ------------------
  324|  17.2k|            obu_end = gb.ptr + len;
  325|  17.2k|        }
  326|       |
  327|  28.5k|        if (type == DAV1D_OBU_SEQ_HDR) {
  ------------------
  |  Branch (327:13): [True: 12.9k, False: 15.5k]
  ------------------
  328|  12.9k|            if ((res = parse_seq_hdr(out, &gb, 0)) < 0) return res;
  ------------------
  |  Branch (328:17): [True: 2.07k, False: 10.8k]
  ------------------
  329|  10.8k|            if (gb.ptr > obu_end) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    301|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (329:17): [True: 301, False: 10.5k]
  ------------------
  330|  10.5k|            dav1d_bytealign_get_bits(&gb);
  331|  10.5k|        }
  332|       |
  333|  26.1k|        if (gb.error) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    393|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (333:13): [True: 393, False: 25.7k]
  ------------------
  334|  26.1k|        assert(gb.state == 0 && gb.bits_left == 0);
  ------------------
  |  Branch (334:9): [True: 25.7k, False: 0]
  |  Branch (334:9): [True: 25.7k, False: 0]
  ------------------
  335|  25.7k|        gb.ptr = obu_end;
  336|  25.7k|    } while (gb.ptr < gb.ptr_end);
  ------------------
  |  Branch (336:14): [True: 15.6k, False: 10.1k]
  ------------------
  337|       |
  338|  10.1k|    return res;
  339|  13.1k|}
dav1d_parse_obus:
 1169|   114k|ptrdiff_t dav1d_parse_obus(Dav1dContext *const c, Dav1dData *const in) {
 1170|   114k|    GetBits gb;
 1171|   114k|    int res;
 1172|       |
 1173|   114k|    dav1d_init_get_bits(&gb, in->data, in->sz);
 1174|       |
 1175|       |    // obu header
 1176|   114k|    const int obu_forbidden_bit = dav1d_get_bit(&gb);
 1177|   114k|    if (c->strict_std_compliance && obu_forbidden_bit) goto error;
  ------------------
  |  Branch (1177:9): [True: 0, False: 114k]
  |  Branch (1177:37): [True: 0, False: 0]
  ------------------
 1178|   114k|    const enum Dav1dObuType type = dav1d_get_bits(&gb, 4);
 1179|   114k|    const int has_extension = dav1d_get_bit(&gb);
 1180|   114k|    const int has_length_field = dav1d_get_bit(&gb);
 1181|   114k|    dav1d_get_bit(&gb); // reserved
 1182|       |
 1183|   114k|    int temporal_id = 0, spatial_id = 0;
 1184|   114k|    if (has_extension) {
  ------------------
  |  Branch (1184:9): [True: 5.08k, False: 109k]
  ------------------
 1185|  5.08k|        temporal_id = dav1d_get_bits(&gb, 3);
 1186|  5.08k|        spatial_id = dav1d_get_bits(&gb, 2);
 1187|  5.08k|        dav1d_get_bits(&gb, 3); // reserved
 1188|  5.08k|    }
 1189|       |
 1190|   114k|    if (has_length_field) {
  ------------------
  |  Branch (1190:9): [True: 40.3k, False: 74.0k]
  ------------------
 1191|  40.3k|        const size_t len = dav1d_get_uleb128(&gb);
 1192|  40.3k|        if (len > (size_t)(gb.ptr_end - gb.ptr)) goto error;
  ------------------
  |  Branch (1192:13): [True: 847, False: 39.4k]
  ------------------
 1193|  39.4k|        gb.ptr_end = gb.ptr + len;
 1194|  39.4k|    }
 1195|   113k|    if (gb.error) goto error;
  ------------------
  |  Branch (1195:9): [True: 456, False: 113k]
  ------------------
 1196|       |
 1197|       |    // We must have read a whole number of bytes at this point (1 byte
 1198|       |    // for the header and whole bytes at a time when reading the
 1199|       |    // leb128 length field).
 1200|   113k|    assert(gb.bits_left == 0);
  ------------------
  |  Branch (1200:5): [True: 113k, False: 0]
  ------------------
 1201|       |
 1202|       |    // skip obu not belonging to the selected temporal/spatial layer
 1203|   113k|    if (type != DAV1D_OBU_SEQ_HDR && type != DAV1D_OBU_TD &&
  ------------------
  |  Branch (1203:9): [True: 88.5k, False: 24.5k]
  |  Branch (1203:38): [True: 85.7k, False: 2.78k]
  ------------------
 1204|  85.7k|        has_extension && c->operating_point_idc != 0)
  ------------------
  |  Branch (1204:9): [True: 3.72k, False: 82.0k]
  |  Branch (1204:26): [True: 1.20k, False: 2.52k]
  ------------------
 1205|  1.20k|    {
 1206|  1.20k|        const int in_temporal_layer = (c->operating_point_idc >> temporal_id) & 1;
 1207|  1.20k|        const int in_spatial_layer = (c->operating_point_idc >> (spatial_id + 8)) & 1;
 1208|  1.20k|        if (!in_temporal_layer || !in_spatial_layer)
  ------------------
  |  Branch (1208:13): [True: 349, False: 852]
  |  Branch (1208:35): [True: 239, False: 613]
  ------------------
 1209|    588|            return gb.ptr_end - gb.ptr_start;
 1210|  1.20k|    }
 1211|       |
 1212|   112k|    switch (type) {
 1213|  24.5k|    case DAV1D_OBU_SEQ_HDR: {
  ------------------
  |  Branch (1213:5): [True: 24.5k, False: 87.9k]
  ------------------
 1214|  24.5k|        Dav1dRef *ref = dav1d_ref_create_using_pool(c->seq_hdr_pool,
 1215|  24.5k|                                                    sizeof(Dav1dSequenceHeader));
 1216|  24.5k|        if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1216:13): [True: 0, False: 24.5k]
  ------------------
 1217|  24.5k|        Dav1dSequenceHeader *seq_hdr = ref->data;
 1218|  24.5k|        if ((res = parse_seq_hdr(seq_hdr, &gb, c->strict_std_compliance)) < 0) {
  ------------------
  |  Branch (1218:13): [True: 3.62k, False: 20.9k]
  ------------------
 1219|  3.62k|            dav1d_log(c, "Error parsing sequence header\n");
  ------------------
  |  |   44|  3.62k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 3.62k]
  |  |  ------------------
  ------------------
 1220|  3.62k|            dav1d_ref_dec(&ref);
 1221|  3.62k|            goto error;
 1222|  3.62k|        }
 1223|       |
 1224|  20.9k|        const int op_idx =
 1225|  20.9k|            c->operating_point < seq_hdr->num_operating_points ? c->operating_point : 0;
  ------------------
  |  Branch (1225:13): [True: 20.9k, False: 0]
  ------------------
 1226|  20.9k|        c->operating_point_idc = seq_hdr->operating_points[op_idx].idc;
 1227|  20.9k|        const unsigned spatial_mask = c->operating_point_idc >> 8;
 1228|  20.9k|        c->max_spatial_id = spatial_mask ? ulog2(spatial_mask) : 0;
  ------------------
  |  Branch (1228:29): [True: 5.61k, False: 15.3k]
  ------------------
 1229|       |
 1230|       |        // If we have read a sequence header which is different from
 1231|       |        // the old one, this is a new video sequence and can't use any
 1232|       |        // previous state. Free that state.
 1233|       |
 1234|  20.9k|        if (!c->seq_hdr) {
  ------------------
  |  Branch (1234:13): [True: 9.83k, False: 11.1k]
  ------------------
 1235|  9.83k|            c->frame_hdr = NULL;
 1236|  9.83k|            c->frame_flags |= PICTURE_FLAG_NEW_SEQUENCE;
 1237|       |        // see 7.5, operating_parameter_info is allowed to change in
 1238|       |        // sequence headers of a single sequence
 1239|  11.1k|        } else if (memcmp(seq_hdr, c->seq_hdr, offsetof(Dav1dSequenceHeader, operating_parameter_info))) {
  ------------------
  |  Branch (1239:20): [True: 5.74k, False: 5.36k]
  ------------------
 1240|  5.74k|            c->frame_hdr = NULL;
 1241|  5.74k|            c->mastering_display = NULL;
 1242|  5.74k|            c->content_light = NULL;
 1243|  5.74k|            dav1d_ref_dec(&c->mastering_display_ref);
 1244|  5.74k|            dav1d_ref_dec(&c->content_light_ref);
 1245|  51.7k|            for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1245:29): [True: 45.9k, False: 5.74k]
  ------------------
 1246|  45.9k|                if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (1246:21): [True: 1.53k, False: 44.4k]
  ------------------
 1247|  1.53k|                    dav1d_thread_picture_unref(&c->refs[i].p);
 1248|  45.9k|                dav1d_ref_dec(&c->refs[i].segmap);
 1249|  45.9k|                dav1d_ref_dec(&c->refs[i].refmvs);
 1250|  45.9k|                dav1d_cdf_thread_unref(&c->cdf[i]);
 1251|  45.9k|            }
 1252|  5.74k|            c->frame_flags |= PICTURE_FLAG_NEW_SEQUENCE;
 1253|       |        // If operating_parameter_info changed, signal it
 1254|  5.74k|        } else if (memcmp(seq_hdr->operating_parameter_info, c->seq_hdr->operating_parameter_info,
  ------------------
  |  Branch (1254:20): [True: 576, False: 4.78k]
  ------------------
 1255|  5.36k|                          sizeof(seq_hdr->operating_parameter_info)))
 1256|    576|        {
 1257|    576|            c->frame_flags |= PICTURE_FLAG_NEW_OP_PARAMS_INFO;
 1258|    576|        }
 1259|  20.9k|        dav1d_ref_dec(&c->seq_hdr_ref);
 1260|  20.9k|        c->seq_hdr_ref = ref;
 1261|  20.9k|        c->seq_hdr = seq_hdr;
 1262|  20.9k|        break;
 1263|  24.5k|    }
 1264|  1.05k|    case DAV1D_OBU_REDUNDANT_FRAME_HDR:
  ------------------
  |  Branch (1264:5): [True: 1.05k, False: 111k]
  ------------------
 1265|  1.05k|        if (c->frame_hdr) break;
  ------------------
  |  Branch (1265:13): [True: 525, False: 533]
  ------------------
 1266|       |        // fall-through
 1267|  62.1k|    case DAV1D_OBU_FRAME:
  ------------------
  |  Branch (1267:5): [True: 61.6k, False: 50.8k]
  ------------------
 1268|  76.2k|    case DAV1D_OBU_FRAME_HDR:
  ------------------
  |  Branch (1268:5): [True: 14.1k, False: 98.3k]
  ------------------
 1269|  76.2k|        if (!c->seq_hdr) goto error;
  ------------------
  |  Branch (1269:13): [True: 194, False: 76.0k]
  ------------------
 1270|  76.0k|        if (!c->frame_hdr_ref) {
  ------------------
  |  Branch (1270:13): [True: 49.9k, False: 26.1k]
  ------------------
 1271|  49.9k|            c->frame_hdr_ref = dav1d_ref_create_using_pool(c->frame_hdr_pool,
 1272|  49.9k|                                                           sizeof(Dav1dFrameHeader));
 1273|  49.9k|            if (!c->frame_hdr_ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1273:17): [True: 0, False: 49.9k]
  ------------------
 1274|  49.9k|        }
 1275|  76.0k|#ifndef NDEBUG
 1276|       |        // ensure that the reference is writable
 1277|  76.0k|        assert(dav1d_ref_is_writable(c->frame_hdr_ref));
  ------------------
  |  Branch (1277:9): [True: 76.0k, False: 0]
  ------------------
 1278|  76.0k|#endif
 1279|  76.0k|        c->frame_hdr = c->frame_hdr_ref->data;
 1280|  76.0k|        memset(c->frame_hdr, 0, sizeof(*c->frame_hdr));
 1281|  76.0k|        c->frame_hdr->temporal_id = temporal_id;
 1282|  76.0k|        c->frame_hdr->spatial_id = spatial_id;
 1283|  76.0k|        if ((res = parse_frame_hdr(c, &gb)) < 0) {
  ------------------
  |  Branch (1283:13): [True: 5.35k, False: 70.6k]
  ------------------
 1284|  5.35k|            c->frame_hdr = NULL;
 1285|  5.35k|            goto error;
 1286|  5.35k|        }
 1287|  73.4k|        for (int n = 0; n < c->n_tile_data; n++)
  ------------------
  |  Branch (1287:25): [True: 2.79k, False: 70.6k]
  ------------------
 1288|  2.79k|            dav1d_data_unref_internal(&c->tile[n].data);
 1289|  70.6k|        c->n_tile_data = 0;
 1290|  70.6k|        c->n_tiles = 0;
 1291|  70.6k|        if (type != DAV1D_OBU_FRAME) {
  ------------------
  |  Branch (1291:13): [True: 13.3k, False: 57.3k]
  ------------------
 1292|       |            // This is actually a frame header OBU so read the
 1293|       |            // trailing bit and check for overrun.
 1294|  13.3k|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1294:17): [True: 5.54k, False: 7.79k]
  ------------------
 1295|  5.54k|                c->frame_hdr = NULL;
 1296|  5.54k|                goto error;
 1297|  5.54k|            }
 1298|  13.3k|        }
 1299|       |
 1300|  65.1k|        if (c->frame_size_limit && (int64_t)c->frame_hdr->width[1] *
  ------------------
  |  Branch (1300:13): [True: 65.1k, False: 0]
  |  Branch (1300:36): [True: 546, False: 64.6k]
  ------------------
 1301|  65.1k|            c->frame_hdr->height > c->frame_size_limit)
 1302|    546|        {
 1303|    546|            dav1d_log(c, "Frame size %dx%d exceeds limit %u\n", c->frame_hdr->width[1],
  ------------------
  |  |   44|    546|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 546]
  |  |  ------------------
  ------------------
 1304|    546|                      c->frame_hdr->height, c->frame_size_limit);
 1305|    546|            c->frame_hdr = NULL;
 1306|    546|            return DAV1D_ERR(ERANGE);
  ------------------
  |  |   58|    546|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1307|    546|        }
 1308|       |
 1309|  64.6k|        if (type != DAV1D_OBU_FRAME)
  ------------------
  |  Branch (1309:13): [True: 7.79k, False: 56.8k]
  ------------------
 1310|  7.79k|            break;
 1311|       |        // OBU_FRAMEs shouldn't be signaled with show_existing_frame
 1312|  56.8k|        if (c->frame_hdr->show_existing_frame) {
  ------------------
  |  Branch (1312:13): [True: 204, False: 56.6k]
  ------------------
 1313|    204|            c->frame_hdr = NULL;
 1314|    204|            goto error;
 1315|    204|        }
 1316|       |
 1317|       |        // This is the frame header at the start of a frame OBU.
 1318|       |        // There's no trailing bit at the end to skip, but we do need
 1319|       |        // to align to the next byte.
 1320|  56.6k|        dav1d_bytealign_get_bits(&gb);
 1321|       |        // fall-through
 1322|  58.5k|    case DAV1D_OBU_TILE_GRP: {
  ------------------
  |  Branch (1322:5): [True: 1.90k, False: 110k]
  ------------------
 1323|  58.5k|        if (!c->frame_hdr) goto error;
  ------------------
  |  Branch (1323:13): [True: 251, False: 58.2k]
  ------------------
 1324|  58.2k|        if (c->n_tile_data_alloc < c->n_tile_data + 1) {
  ------------------
  |  Branch (1324:13): [True: 9.14k, False: 49.1k]
  ------------------
 1325|  9.14k|            if ((c->n_tile_data + 1) > INT_MAX / (int)sizeof(*c->tile)) goto error;
  ------------------
  |  Branch (1325:17): [True: 0, False: 9.14k]
  ------------------
 1326|  9.14k|            struct Dav1dTileGroup *tile = dav1d_realloc(ALLOC_TILE, c->tile,
  ------------------
  |  |  133|  9.14k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
 1327|  9.14k|                                                        (c->n_tile_data + 1) * sizeof(*c->tile));
 1328|  9.14k|            if (!tile) goto error;
  ------------------
  |  Branch (1328:17): [True: 0, False: 9.14k]
  ------------------
 1329|  9.14k|            c->tile = tile;
 1330|  9.14k|            memset(c->tile + c->n_tile_data, 0, sizeof(*c->tile));
 1331|  9.14k|            c->n_tile_data_alloc = c->n_tile_data + 1;
 1332|  9.14k|        }
 1333|  58.2k|        parse_tile_hdr(c, &gb);
 1334|       |        // Align to the next byte boundary and check for overrun.
 1335|  58.2k|        dav1d_bytealign_get_bits(&gb);
 1336|  58.2k|        if (gb.error) goto error;
  ------------------
  |  Branch (1336:13): [True: 8.19k, False: 50.0k]
  ------------------
 1337|       |
 1338|  50.0k|        dav1d_data_ref(&c->tile[c->n_tile_data].data, in);
 1339|  50.0k|        c->tile[c->n_tile_data].data.data = gb.ptr;
 1340|  50.0k|        c->tile[c->n_tile_data].data.sz = (size_t)(gb.ptr_end - gb.ptr);
 1341|       |        // ensure tile groups are in order and sane, see 6.10.1
 1342|  50.0k|        if (c->tile[c->n_tile_data].start > c->tile[c->n_tile_data].end ||
  ------------------
  |  Branch (1342:13): [True: 390, False: 49.6k]
  ------------------
 1343|  49.6k|            c->tile[c->n_tile_data].start != c->n_tiles)
  ------------------
  |  Branch (1343:13): [True: 426, False: 49.2k]
  ------------------
 1344|    816|        {
 1345|  1.91k|            for (int i = 0; i <= c->n_tile_data; i++)
  ------------------
  |  Branch (1345:29): [True: 1.09k, False: 816]
  ------------------
 1346|  1.09k|                dav1d_data_unref_internal(&c->tile[i].data);
 1347|    816|            c->n_tile_data = 0;
 1348|    816|            c->n_tiles = 0;
 1349|    816|            goto error;
 1350|    816|        }
 1351|  49.2k|        c->n_tiles += 1 + c->tile[c->n_tile_data].end -
 1352|  49.2k|                          c->tile[c->n_tile_data].start;
 1353|  49.2k|        c->n_tile_data++;
 1354|  49.2k|        break;
 1355|  50.0k|    }
 1356|  3.06k|    case DAV1D_OBU_METADATA: {
  ------------------
  |  Branch (1356:5): [True: 3.06k, False: 109k]
  ------------------
 1357|  3.06k|#define DEBUG_OBU_METADATA 0
 1358|       |#if DEBUG_OBU_METADATA
 1359|       |        const uint8_t *const init_ptr = gb.ptr;
 1360|       |#endif
 1361|       |        // obu metadta type field
 1362|  3.06k|        const enum ObuMetaType meta_type = dav1d_get_uleb128(&gb);
 1363|  3.06k|        if (gb.error) goto error;
  ------------------
  |  Branch (1363:13): [True: 278, False: 2.79k]
  ------------------
 1364|       |
 1365|  2.79k|        switch (meta_type) {
 1366|    639|        case OBU_META_HDR_CLL: {
  ------------------
  |  Branch (1366:9): [True: 639, False: 2.15k]
  ------------------
 1367|    639|            Dav1dRef *ref = dav1d_ref_create(ALLOC_OBU_META,
  ------------------
  |  |   49|    639|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
 1368|    639|                                             sizeof(Dav1dContentLightLevel));
 1369|    639|            if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1369:17): [True: 0, False: 639]
  ------------------
 1370|    639|            Dav1dContentLightLevel *const content_light = ref->data;
 1371|       |
 1372|    639|            content_light->max_content_light_level = dav1d_get_bits(&gb, 16);
 1373|       |#if DEBUG_OBU_METADATA
 1374|       |            printf("CLLOBU: max-content-light-level: %d [off=%td]\n",
 1375|       |                   content_light->max_content_light_level,
 1376|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1377|       |#endif
 1378|    639|            content_light->max_frame_average_light_level = dav1d_get_bits(&gb, 16);
 1379|       |#if DEBUG_OBU_METADATA
 1380|       |            printf("CLLOBU: max-frame-average-light-level: %d [off=%td]\n",
 1381|       |                   content_light->max_frame_average_light_level,
 1382|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1383|       |#endif
 1384|       |
 1385|    639|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1385:17): [True: 411, False: 228]
  ------------------
 1386|    411|                dav1d_ref_dec(&ref);
 1387|    411|                goto error;
 1388|    411|            }
 1389|       |
 1390|    228|            dav1d_ref_dec(&c->content_light_ref);
 1391|    228|            c->content_light = content_light;
 1392|    228|            c->content_light_ref = ref;
 1393|    228|            break;
 1394|    639|        }
 1395|    473|        case OBU_META_HDR_MDCV: {
  ------------------
  |  Branch (1395:9): [True: 473, False: 2.31k]
  ------------------
 1396|    473|            Dav1dRef *ref = dav1d_ref_create(ALLOC_OBU_META,
  ------------------
  |  |   49|    473|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
 1397|    473|                                             sizeof(Dav1dMasteringDisplay));
 1398|    473|            if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1398:17): [True: 0, False: 473]
  ------------------
 1399|    473|            Dav1dMasteringDisplay *const mastering_display = ref->data;
 1400|       |
 1401|  1.89k|            for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (1401:29): [True: 1.41k, False: 473]
  ------------------
 1402|  1.41k|                mastering_display->primaries[i][0] = dav1d_get_bits(&gb, 16);
 1403|  1.41k|                mastering_display->primaries[i][1] = dav1d_get_bits(&gb, 16);
 1404|       |#if DEBUG_OBU_METADATA
 1405|       |                printf("MDCVOBU: primaries[%d]: (%d, %d) [off=%td]\n", i,
 1406|       |                       mastering_display->primaries[i][0],
 1407|       |                       mastering_display->primaries[i][1],
 1408|       |                       (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1409|       |#endif
 1410|  1.41k|            }
 1411|    473|            mastering_display->white_point[0] = dav1d_get_bits(&gb, 16);
 1412|       |#if DEBUG_OBU_METADATA
 1413|       |            printf("MDCVOBU: white-point-x: %d [off=%td]\n",
 1414|       |                   mastering_display->white_point[0],
 1415|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1416|       |#endif
 1417|    473|            mastering_display->white_point[1] = dav1d_get_bits(&gb, 16);
 1418|       |#if DEBUG_OBU_METADATA
 1419|       |            printf("MDCVOBU: white-point-y: %d [off=%td]\n",
 1420|       |                   mastering_display->white_point[1],
 1421|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1422|       |#endif
 1423|    473|            mastering_display->max_luminance = dav1d_get_bits(&gb, 32);
 1424|       |#if DEBUG_OBU_METADATA
 1425|       |            printf("MDCVOBU: max-luminance: %d [off=%td]\n",
 1426|       |                   mastering_display->max_luminance,
 1427|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1428|       |#endif
 1429|    473|            mastering_display->min_luminance = dav1d_get_bits(&gb, 32);
 1430|       |#if DEBUG_OBU_METADATA
 1431|       |            printf("MDCVOBU: min-luminance: %d [off=%td]\n",
 1432|       |                   mastering_display->min_luminance,
 1433|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1434|       |#endif
 1435|    473|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1435:17): [True: 278, False: 195]
  ------------------
 1436|    278|                dav1d_ref_dec(&ref);
 1437|    278|                goto error;
 1438|    278|            }
 1439|       |
 1440|    195|            dav1d_ref_dec(&c->mastering_display_ref);
 1441|    195|            c->mastering_display = mastering_display;
 1442|    195|            c->mastering_display_ref = ref;
 1443|    195|            break;
 1444|    473|        }
 1445|  1.18k|        case OBU_META_ITUT_T35: {
  ------------------
  |  Branch (1445:9): [True: 1.18k, False: 1.60k]
  ------------------
 1446|  1.18k|            ptrdiff_t payload_size = gb.ptr_end - gb.ptr;
 1447|       |            // Don't take into account all the trailing bits for payload_size
 1448|  1.61k|            while (payload_size > 0 && !gb.ptr[payload_size - 1])
  ------------------
  |  Branch (1448:20): [True: 1.34k, False: 269]
  |  Branch (1448:40): [True: 436, False: 913]
  ------------------
 1449|    436|                payload_size--; // trailing_zero_bit x 8
 1450|  1.18k|            payload_size--; // trailing_one_bit + trailing_zero_bit x 7
 1451|       |
 1452|  1.18k|            int country_code_extension_byte = 0;
 1453|  1.18k|            const int country_code = dav1d_get_bits(&gb, 8);
 1454|  1.18k|            payload_size--;
 1455|  1.18k|            if (country_code == 0xFF) {
  ------------------
  |  Branch (1455:17): [True: 434, False: 748]
  ------------------
 1456|    434|                country_code_extension_byte = dav1d_get_bits(&gb, 8);
 1457|    434|                payload_size--;
 1458|    434|            }
 1459|       |
 1460|  1.18k|            if (payload_size <= 0 || gb.ptr[payload_size] != 0x80) {
  ------------------
  |  Branch (1460:17): [True: 274, False: 908]
  |  Branch (1460:38): [True: 241, False: 667]
  ------------------
 1461|    515|                dav1d_log(c, "Malformed ITU-T T.35 metadata message format\n");
  ------------------
  |  |   44|    515|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 515]
  |  |  ------------------
  ------------------
 1462|    515|                break;
 1463|    515|            }
 1464|       |
 1465|    667|            if ((c->n_itut_t35 + 1) > INT_MAX / (int)sizeof(*c->itut_t35)) goto error;
  ------------------
  |  Branch (1465:17): [True: 0, False: 667]
  ------------------
 1466|    667|            struct Dav1dITUTT35 *itut_t35 = dav1d_realloc(ALLOC_OBU_META, c->itut_t35,
  ------------------
  |  |  133|    667|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
 1467|    667|                                                          (c->n_itut_t35 + 1) * sizeof(*c->itut_t35));
 1468|    667|            if (!itut_t35) goto error;
  ------------------
  |  Branch (1468:17): [True: 0, False: 667]
  ------------------
 1469|    667|            c->itut_t35 = itut_t35;
 1470|    667|            memset(c->itut_t35 + c->n_itut_t35, 0, sizeof(*c->itut_t35));
 1471|       |
 1472|    667|            struct itut_t35_ctx_context *itut_t35_ctx;
 1473|    667|            if (!c->n_itut_t35) {
  ------------------
  |  Branch (1473:17): [True: 297, False: 370]
  ------------------
 1474|    297|                assert(!c->itut_t35_ref);
  ------------------
  |  Branch (1474:17): [True: 297, False: 0]
  ------------------
 1475|    297|                itut_t35_ctx = dav1d_malloc(ALLOC_OBU_META, sizeof(struct itut_t35_ctx_context));
  ------------------
  |  |  132|    297|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 1476|    297|                if (!itut_t35_ctx) goto error;
  ------------------
  |  Branch (1476:21): [True: 0, False: 297]
  ------------------
 1477|    297|                c->itut_t35_ref = dav1d_ref_init(&itut_t35_ctx->ref, c->itut_t35,
 1478|    297|                                                 dav1d_picture_free_itut_t35, itut_t35_ctx, 0);
 1479|    370|            } else {
 1480|    370|                assert(c->itut_t35_ref && atomic_load(&c->itut_t35_ref->ref_cnt) == 1);
  ------------------
  |  Branch (1480:17): [True: 370, False: 0]
  |  Branch (1480:17): [True: 370, False: 0]
  ------------------
 1481|    370|                itut_t35_ctx = c->itut_t35_ref->user_data;
 1482|    370|                c->itut_t35_ref->const_data = (uint8_t *)c->itut_t35;
 1483|    370|            }
 1484|    667|            itut_t35_ctx->itut_t35 = c->itut_t35;
 1485|    667|            itut_t35_ctx->n_itut_t35 = c->n_itut_t35 + 1;
 1486|       |
 1487|    667|            Dav1dITUTT35 *const itut_t35_metadata = &c->itut_t35[c->n_itut_t35];
 1488|    667|            itut_t35_metadata->payload = dav1d_malloc(ALLOC_OBU_META, payload_size);
  ------------------
  |  |  132|    667|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 1489|    667|            if (!itut_t35_metadata->payload) goto error;
  ------------------
  |  Branch (1489:17): [True: 0, False: 667]
  ------------------
 1490|       |
 1491|    667|            itut_t35_metadata->country_code = country_code;
 1492|    667|            itut_t35_metadata->country_code_extension_byte = country_code_extension_byte;
 1493|    667|            itut_t35_metadata->payload_size = payload_size;
 1494|       |
 1495|       |            // We know that we've read a whole number of bytes and that the
 1496|       |            // payload is within the OBU boundaries, so just use memcpy()
 1497|    667|            assert(gb.bits_left == 0);
  ------------------
  |  Branch (1497:13): [True: 667, False: 0]
  ------------------
 1498|    667|            memcpy(itut_t35_metadata->payload, gb.ptr, payload_size);
 1499|       |
 1500|    667|            c->n_itut_t35++;
 1501|    667|            break;
 1502|    667|        }
 1503|      1|        case OBU_META_SCALABILITY:
  ------------------
  |  Branch (1503:9): [True: 1, False: 2.78k]
  ------------------
 1504|      1|        case OBU_META_TIMECODE:
  ------------------
  |  Branch (1504:9): [True: 0, False: 2.79k]
  ------------------
 1505|       |            // ignore metadata OBUs we don't care about
 1506|      1|            break;
 1507|    495|        default:
  ------------------
  |  Branch (1507:9): [True: 495, False: 2.29k]
  ------------------
 1508|       |            // print a warning but don't fail for unknown types
 1509|    495|            if (meta_type > 31) // Types 6 to 31 are "Unregistered user private", so ignore them.
  ------------------
  |  Branch (1509:17): [True: 256, False: 239]
  ------------------
 1510|    256|                dav1d_log(c, "Unknown Metadata OBU type %d\n", meta_type);
  ------------------
  |  |   44|    256|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 256]
  |  |  ------------------
  ------------------
 1511|    495|            break;
 1512|  2.79k|        }
 1513|       |
 1514|  2.10k|        break;
 1515|  2.79k|    }
 1516|  2.78k|    case DAV1D_OBU_TD:
  ------------------
  |  Branch (1516:5): [True: 2.78k, False: 109k]
  ------------------
 1517|  2.78k|        c->frame_flags |= PICTURE_FLAG_NEW_TEMPORAL_UNIT;
 1518|  2.78k|        break;
 1519|    349|    case DAV1D_OBU_PADDING:
  ------------------
  |  Branch (1519:5): [True: 349, False: 112k]
  ------------------
 1520|       |        // ignore OBUs we don't care about
 1521|    349|        break;
 1522|  3.06k|    default:
  ------------------
  |  Branch (1522:5): [True: 3.06k, False: 109k]
  ------------------
 1523|       |        // print a warning but don't fail for unknown types
 1524|  3.06k|        dav1d_log(c, "Unknown OBU type %d of size %td\n", type, gb.ptr_end - gb.ptr);
  ------------------
  |  |   44|  3.06k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 3.06k]
  |  |  ------------------
  ------------------
 1525|  3.06k|        break;
 1526|   112k|    }
 1527|       |
 1528|  86.8k|    if (c->seq_hdr && c->frame_hdr) {
  ------------------
  |  Branch (1528:9): [True: 85.6k, False: 1.18k]
  |  Branch (1528:23): [True: 59.0k, False: 26.6k]
  ------------------
 1529|  59.0k|        if (c->frame_hdr->show_existing_frame) {
  ------------------
  |  Branch (1529:13): [True: 5.03k, False: 53.9k]
  ------------------
 1530|  5.03k|            if (!c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1530:17): [True: 218, False: 4.81k]
  ------------------
 1531|  4.81k|            switch (c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr->frame_type) {
 1532|    227|            case DAV1D_FRAME_TYPE_INTER:
  ------------------
  |  Branch (1532:13): [True: 227, False: 4.58k]
  ------------------
 1533|    299|            case DAV1D_FRAME_TYPE_SWITCH:
  ------------------
  |  Branch (1533:13): [True: 72, False: 4.74k]
  ------------------
 1534|    299|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_REFERENCE)
  ------------------
  |  Branch (1534:21): [True: 0, False: 299]
  ------------------
 1535|      0|                    goto skip;
 1536|    299|                break;
 1537|    299|            case DAV1D_FRAME_TYPE_INTRA:
  ------------------
  |  Branch (1537:13): [True: 201, False: 4.61k]
  ------------------
 1538|    201|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_INTRA)
  ------------------
  |  Branch (1538:21): [True: 0, False: 201]
  ------------------
 1539|      0|                    goto skip;
 1540|       |                // fall-through
 1541|  4.51k|            default:
  ------------------
  |  Branch (1541:13): [True: 4.31k, False: 500]
  ------------------
 1542|  4.51k|                break;
 1543|  4.81k|            }
 1544|  4.81k|            if (!c->refs[c->frame_hdr->existing_frame_idx].p.p.data[0]) goto error;
  ------------------
  |  Branch (1544:17): [True: 0, False: 4.81k]
  ------------------
 1545|  4.81k|            if (c->strict_std_compliance &&
  ------------------
  |  Branch (1545:17): [True: 0, False: 4.81k]
  ------------------
 1546|      0|                !c->refs[c->frame_hdr->existing_frame_idx].p.showable)
  ------------------
  |  Branch (1546:17): [True: 0, False: 0]
  ------------------
 1547|      0|            {
 1548|      0|                goto error;
 1549|      0|            }
 1550|  4.81k|            if (c->n_fc == 1) {
  ------------------
  |  Branch (1550:17): [True: 4.81k, False: 0]
  ------------------
 1551|  4.81k|                dav1d_thread_picture_ref(&c->out,
 1552|  4.81k|                                         &c->refs[c->frame_hdr->existing_frame_idx].p);
 1553|  4.81k|                dav1d_picture_copy_props(&c->out.p,
 1554|  4.81k|                                         c->content_light, c->content_light_ref,
 1555|  4.81k|                                         c->mastering_display, c->mastering_display_ref,
 1556|  4.81k|                                         c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
 1557|  4.81k|                                         &in->m);
 1558|       |                // Must be removed from the context after being attached to the frame
 1559|  4.81k|                dav1d_ref_dec(&c->itut_t35_ref);
 1560|  4.81k|                c->itut_t35 = NULL;
 1561|  4.81k|                c->n_itut_t35 = 0;
 1562|  4.81k|                c->event_flags |= dav1d_picture_get_event_flags(&c->refs[c->frame_hdr->existing_frame_idx].p);
 1563|  4.81k|            } else {
 1564|      0|                pthread_mutex_lock(&c->task_thread.lock);
 1565|       |                // need to append this to the frame output queue
 1566|      0|                const unsigned next = c->frame_thread.next++;
 1567|      0|                if (c->frame_thread.next == c->n_fc)
  ------------------
  |  Branch (1567:21): [True: 0, False: 0]
  ------------------
 1568|      0|                    c->frame_thread.next = 0;
 1569|       |
 1570|      0|                Dav1dFrameContext *const f = &c->fc[next];
 1571|      0|                while (f->n_tile_data > 0)
  ------------------
  |  Branch (1571:24): [True: 0, False: 0]
  ------------------
 1572|      0|                    pthread_cond_wait(&f->task_thread.cond,
 1573|      0|                                      &f->task_thread.ttd->lock);
 1574|      0|                Dav1dThreadPicture *const out_delayed =
 1575|      0|                    &c->frame_thread.out_delayed[next];
 1576|      0|                if (out_delayed->p.data[0] || atomic_load(&f->task_thread.error)) {
  ------------------
  |  Branch (1576:21): [True: 0, False: 0]
  |  Branch (1576:47): [True: 0, False: 0]
  ------------------
 1577|      0|                    unsigned first = atomic_load(&c->task_thread.first);
 1578|      0|                    if (first + 1U < c->n_fc)
  ------------------
  |  Branch (1578:25): [True: 0, False: 0]
  ------------------
 1579|      0|                        atomic_fetch_add(&c->task_thread.first, 1U);
 1580|      0|                    else
 1581|      0|                        atomic_store(&c->task_thread.first, 0);
 1582|      0|                    atomic_compare_exchange_strong(&c->task_thread.reset_task_cur,
 1583|      0|                                                   &first, UINT_MAX);
 1584|      0|                    if (c->task_thread.cur && c->task_thread.cur < c->n_fc)
  ------------------
  |  Branch (1584:25): [True: 0, False: 0]
  |  Branch (1584:47): [True: 0, False: 0]
  ------------------
 1585|      0|                        c->task_thread.cur--;
 1586|      0|                }
 1587|      0|                const int error = f->task_thread.retval;
 1588|      0|                if (error) {
  ------------------
  |  Branch (1588:21): [True: 0, False: 0]
  ------------------
 1589|      0|                    c->cached_error = error;
 1590|      0|                    f->task_thread.retval = 0;
 1591|      0|                    dav1d_data_props_copy(&c->cached_error_props, &out_delayed->p.m);
 1592|      0|                    dav1d_thread_picture_unref(out_delayed);
 1593|      0|                } else if (out_delayed->p.data[0]) {
  ------------------
  |  Branch (1593:28): [True: 0, False: 0]
  ------------------
 1594|      0|                    const unsigned progress = atomic_load_explicit(&out_delayed->progress[1],
 1595|      0|                                                                   memory_order_relaxed);
 1596|      0|                    if ((out_delayed->visible || c->output_invisible_frames) &&
  ------------------
  |  Branch (1596:26): [True: 0, False: 0]
  |  Branch (1596:50): [True: 0, False: 0]
  ------------------
 1597|      0|                        progress != FRAME_ERROR)
  ------------------
  |  |   35|      0|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (1597:25): [True: 0, False: 0]
  ------------------
 1598|      0|                    {
 1599|      0|                        dav1d_thread_picture_ref(&c->out, out_delayed);
 1600|      0|                        c->event_flags |= dav1d_picture_get_event_flags(out_delayed);
 1601|      0|                    }
 1602|      0|                    dav1d_thread_picture_unref(out_delayed);
 1603|      0|                }
 1604|      0|                dav1d_thread_picture_ref(out_delayed,
 1605|      0|                                         &c->refs[c->frame_hdr->existing_frame_idx].p);
 1606|      0|                out_delayed->visible = 1;
 1607|      0|                dav1d_picture_copy_props(&out_delayed->p,
 1608|      0|                                         c->content_light, c->content_light_ref,
 1609|      0|                                         c->mastering_display, c->mastering_display_ref,
 1610|      0|                                         c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
 1611|      0|                                         &in->m);
 1612|       |                // Must be removed from the context after being attached to the frame
 1613|      0|                dav1d_ref_dec(&c->itut_t35_ref);
 1614|      0|                c->itut_t35 = NULL;
 1615|      0|                c->n_itut_t35 = 0;
 1616|       |
 1617|      0|                pthread_mutex_unlock(&c->task_thread.lock);
 1618|      0|            }
 1619|  4.81k|            if (c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr->frame_type == DAV1D_FRAME_TYPE_KEY) {
  ------------------
  |  Branch (1619:17): [True: 4.31k, False: 500]
  ------------------
 1620|  4.31k|                const int r = c->frame_hdr->existing_frame_idx;
 1621|  4.31k|                c->refs[r].p.showable = 0;
 1622|  38.8k|                for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1622:33): [True: 34.5k, False: 4.31k]
  ------------------
 1623|  34.5k|                    if (i == r) continue;
  ------------------
  |  Branch (1623:25): [True: 4.31k, False: 30.1k]
  ------------------
 1624|       |
 1625|  30.1k|                    if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (1625:25): [True: 29.8k, False: 341]
  ------------------
 1626|  29.8k|                        dav1d_thread_picture_unref(&c->refs[i].p);
 1627|  30.1k|                    dav1d_thread_picture_ref(&c->refs[i].p, &c->refs[r].p);
 1628|       |
 1629|  30.1k|                    dav1d_cdf_thread_unref(&c->cdf[i]);
 1630|  30.1k|                    dav1d_cdf_thread_ref(&c->cdf[i], &c->cdf[r]);
 1631|       |
 1632|  30.1k|                    dav1d_ref_dec(&c->refs[i].segmap);
 1633|  30.1k|                    c->refs[i].segmap = c->refs[r].segmap;
 1634|  30.1k|                    if (c->refs[r].segmap)
  ------------------
  |  Branch (1634:25): [True: 2.54k, False: 27.6k]
  ------------------
 1635|  2.54k|                        dav1d_ref_inc(c->refs[r].segmap);
 1636|  30.1k|                    dav1d_ref_dec(&c->refs[i].refmvs);
 1637|  30.1k|                }
 1638|  4.31k|            }
 1639|  4.81k|            c->frame_hdr = NULL;
 1640|  53.9k|        } else if (c->n_tiles == c->frame_hdr->tiling.cols * c->frame_hdr->tiling.rows) {
  ------------------
  |  Branch (1640:20): [True: 48.4k, False: 5.51k]
  ------------------
 1641|  48.4k|            switch (c->frame_hdr->frame_type) {
 1642|  15.6k|            case DAV1D_FRAME_TYPE_INTER:
  ------------------
  |  Branch (1642:13): [True: 15.6k, False: 32.8k]
  ------------------
 1643|  16.1k|            case DAV1D_FRAME_TYPE_SWITCH:
  ------------------
  |  Branch (1643:13): [True: 467, False: 47.9k]
  ------------------
 1644|  16.1k|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_REFERENCE ||
  ------------------
  |  Branch (1644:21): [True: 0, False: 16.1k]
  ------------------
 1645|  16.1k|                    (c->decode_frame_type == DAV1D_DECODEFRAMETYPE_REFERENCE &&
  ------------------
  |  Branch (1645:22): [True: 0, False: 16.1k]
  ------------------
 1646|      0|                     !c->frame_hdr->refresh_frame_flags))
  ------------------
  |  Branch (1646:22): [True: 0, False: 0]
  ------------------
 1647|      0|                    goto skip;
 1648|  16.1k|                break;
 1649|  16.1k|            case DAV1D_FRAME_TYPE_INTRA:
  ------------------
  |  Branch (1649:13): [True: 437, False: 48.0k]
  ------------------
 1650|    437|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_INTRA ||
  ------------------
  |  Branch (1650:21): [True: 0, False: 437]
  ------------------
 1651|    437|                    (c->decode_frame_type == DAV1D_DECODEFRAMETYPE_REFERENCE &&
  ------------------
  |  Branch (1651:22): [True: 0, False: 437]
  ------------------
 1652|      0|                     !c->frame_hdr->refresh_frame_flags))
  ------------------
  |  Branch (1652:22): [True: 0, False: 0]
  ------------------
 1653|      0|                    goto skip;
 1654|       |                // fall-through
 1655|  32.3k|            default:
  ------------------
  |  Branch (1655:13): [True: 31.9k, False: 16.5k]
  ------------------
 1656|  32.3k|                break;
 1657|  48.4k|            }
 1658|  48.4k|            if (!c->n_tile_data)
  ------------------
  |  Branch (1658:17): [True: 0, False: 48.4k]
  ------------------
 1659|      0|                goto error;
 1660|  48.4k|            if ((res = dav1d_submit_frame(c)) < 0)
  ------------------
  |  Branch (1660:17): [True: 28.3k, False: 20.0k]
  ------------------
 1661|  28.3k|                return res;
 1662|  48.4k|            assert(!c->n_tile_data);
  ------------------
  |  Branch (1662:13): [True: 20.0k, False: 0]
  ------------------
 1663|  20.0k|            c->frame_hdr = NULL;
 1664|  20.0k|            c->n_tiles = 0;
 1665|  20.0k|        }
 1666|  59.0k|    }
 1667|       |
 1668|  58.2k|    return gb.ptr_end - gb.ptr_start;
 1669|       |
 1670|      0|skip:
 1671|       |    // update refs with only the headers in case we skip the frame
 1672|      0|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1672:21): [True: 0, False: 0]
  ------------------
 1673|      0|        if (c->frame_hdr->refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (1673:13): [True: 0, False: 0]
  ------------------
 1674|      0|            dav1d_thread_picture_unref(&c->refs[i].p);
 1675|      0|            c->refs[i].p.p.frame_hdr = c->frame_hdr;
 1676|      0|            c->refs[i].p.p.seq_hdr = c->seq_hdr;
 1677|      0|            c->refs[i].p.p.frame_hdr_ref = c->frame_hdr_ref;
 1678|      0|            c->refs[i].p.p.seq_hdr_ref = c->seq_hdr_ref;
 1679|      0|            dav1d_ref_inc(c->frame_hdr_ref);
 1680|      0|            dav1d_ref_inc(c->seq_hdr_ref);
 1681|      0|        }
 1682|      0|    }
 1683|       |
 1684|      0|    dav1d_ref_dec(&c->frame_hdr_ref);
 1685|      0|    c->frame_hdr = NULL;
 1686|      0|    c->n_tiles = 0;
 1687|       |
 1688|      0|    return gb.ptr_end - gb.ptr_start;
 1689|       |
 1690|  26.6k|error:
 1691|  26.6k|    dav1d_data_props_copy(&c->cached_error_props, &in->m);
 1692|  26.6k|    dav1d_log(c, gb.error ? "Overrun in OBU bit buffer\n" :
  ------------------
  |  |   44|  26.6k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 26.6k]
  |  |  ------------------
  ------------------
 1693|  26.6k|                            "Error parsing OBU data\n");
 1694|  26.6k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  26.6k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1695|  86.8k|}
obu.c:parse_seq_hdr:
   75|  37.4k|{
   76|  37.4k|#define DEBUG_SEQ_HDR 0
   77|       |
   78|       |#if DEBUG_SEQ_HDR
   79|       |    const unsigned init_bit_pos = dav1d_get_bits_pos(gb);
   80|       |#endif
   81|       |
   82|  37.4k|    memset(hdr, 0, sizeof(*hdr));
   83|  37.4k|    hdr->profile = dav1d_get_bits(gb, 3);
   84|  37.4k|    if (hdr->profile > 2) goto error;
  ------------------
  |  Branch (84:9): [True: 614, False: 36.8k]
  ------------------
   85|       |#if DEBUG_SEQ_HDR
   86|       |    printf("SEQHDR: post-profile: off=%u\n",
   87|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
   88|       |#endif
   89|       |
   90|  36.8k|    hdr->still_picture = dav1d_get_bit(gb);
   91|  36.8k|    hdr->reduced_still_picture_header = dav1d_get_bit(gb);
   92|  36.8k|    if (hdr->reduced_still_picture_header && !hdr->still_picture) goto error;
  ------------------
  |  Branch (92:9): [True: 24.0k, False: 12.8k]
  |  Branch (92:46): [True: 259, False: 23.7k]
  ------------------
   93|       |#if DEBUG_SEQ_HDR
   94|       |    printf("SEQHDR: post-stillpicture_flags: off=%u\n",
   95|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
   96|       |#endif
   97|       |
   98|  36.6k|    if (hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (98:9): [True: 23.7k, False: 12.8k]
  ------------------
   99|  23.7k|        hdr->num_operating_points = 1;
  100|  23.7k|        hdr->operating_points[0].major_level = dav1d_get_bits(gb, 3);
  101|  23.7k|        hdr->operating_points[0].minor_level = dav1d_get_bits(gb, 2);
  102|  23.7k|        hdr->operating_points[0].initial_display_delay = 10;
  103|  23.7k|    } else {
  104|  12.8k|        hdr->timing_info_present = dav1d_get_bit(gb);
  105|  12.8k|        if (hdr->timing_info_present) {
  ------------------
  |  Branch (105:13): [True: 2.31k, False: 10.5k]
  ------------------
  106|  2.31k|            hdr->num_units_in_tick = dav1d_get_bits(gb, 32);
  107|  2.31k|            hdr->time_scale = dav1d_get_bits(gb, 32);
  108|  2.31k|            if (strict_std_compliance && (!hdr->num_units_in_tick || !hdr->time_scale))
  ------------------
  |  Branch (108:17): [True: 0, False: 2.31k]
  |  Branch (108:43): [True: 0, False: 0]
  |  Branch (108:70): [True: 0, False: 0]
  ------------------
  109|      0|                goto error;
  110|  2.31k|            hdr->equal_picture_interval = dav1d_get_bit(gb);
  111|  2.31k|            if (hdr->equal_picture_interval) {
  ------------------
  |  Branch (111:17): [True: 824, False: 1.49k]
  ------------------
  112|    824|                const unsigned num_ticks_per_picture = dav1d_get_vlc(gb);
  113|    824|                if (num_ticks_per_picture == UINT32_MAX)
  ------------------
  |  Branch (113:21): [True: 199, False: 625]
  ------------------
  114|    199|                    goto error;
  115|    625|                hdr->num_ticks_per_picture = num_ticks_per_picture + 1;
  116|    625|            }
  117|       |
  118|  2.11k|            hdr->decoder_model_info_present = dav1d_get_bit(gb);
  119|  2.11k|            if (hdr->decoder_model_info_present) {
  ------------------
  |  Branch (119:17): [True: 1.53k, False: 584]
  ------------------
  120|  1.53k|                hdr->encoder_decoder_buffer_delay_length = dav1d_get_bits(gb, 5) + 1;
  121|  1.53k|                hdr->num_units_in_decoding_tick = dav1d_get_bits(gb, 32);
  122|  1.53k|                if (strict_std_compliance && !hdr->num_units_in_decoding_tick)
  ------------------
  |  Branch (122:21): [True: 0, False: 1.53k]
  |  Branch (122:46): [True: 0, False: 0]
  ------------------
  123|      0|                    goto error;
  124|  1.53k|                hdr->buffer_removal_delay_length = dav1d_get_bits(gb, 5) + 1;
  125|  1.53k|                hdr->frame_presentation_delay_length = dav1d_get_bits(gb, 5) + 1;
  126|  1.53k|            }
  127|  2.11k|        }
  128|       |#if DEBUG_SEQ_HDR
  129|       |        printf("SEQHDR: post-timinginfo: off=%u\n",
  130|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  131|       |#endif
  132|       |
  133|  12.6k|        hdr->display_model_info_present = dav1d_get_bit(gb);
  134|  12.6k|        hdr->num_operating_points = dav1d_get_bits(gb, 5) + 1;
  135|  31.4k|        for (int i = 0; i < hdr->num_operating_points; i++) {
  ------------------
  |  Branch (135:25): [True: 19.4k, False: 12.0k]
  ------------------
  136|  19.4k|            struct Dav1dSequenceHeaderOperatingPoint *const op =
  137|  19.4k|                &hdr->operating_points[i];
  138|  19.4k|            op->idc = dav1d_get_bits(gb, 12);
  139|  19.4k|            if (op->idc && (!(op->idc & 0xff) || !(op->idc & 0xf00)))
  ------------------
  |  Branch (139:17): [True: 14.5k, False: 4.86k]
  |  Branch (139:29): [True: 223, False: 14.3k]
  |  Branch (139:50): [True: 387, False: 13.9k]
  ------------------
  140|    610|                goto error;
  141|  18.8k|            op->major_level = 2 + dav1d_get_bits(gb, 3);
  142|  18.8k|            op->minor_level = dav1d_get_bits(gb, 2);
  143|  18.8k|            if (op->major_level > 3)
  ------------------
  |  Branch (143:17): [True: 6.11k, False: 12.6k]
  ------------------
  144|  6.11k|                op->tier = dav1d_get_bit(gb);
  145|  18.8k|            if (hdr->decoder_model_info_present) {
  ------------------
  |  Branch (145:17): [True: 6.44k, False: 12.3k]
  ------------------
  146|  6.44k|                op->decoder_model_param_present = dav1d_get_bit(gb);
  147|  6.44k|                if (op->decoder_model_param_present) {
  ------------------
  |  Branch (147:21): [True: 5.10k, False: 1.33k]
  ------------------
  148|  5.10k|                    struct Dav1dSequenceHeaderOperatingParameterInfo *const opi =
  149|  5.10k|                        &hdr->operating_parameter_info[i];
  150|  5.10k|                    opi->decoder_buffer_delay =
  151|  5.10k|                        dav1d_get_bits(gb, hdr->encoder_decoder_buffer_delay_length);
  152|  5.10k|                    opi->encoder_buffer_delay =
  153|  5.10k|                        dav1d_get_bits(gb, hdr->encoder_decoder_buffer_delay_length);
  154|  5.10k|                    opi->low_delay_mode = dav1d_get_bit(gb);
  155|  5.10k|                }
  156|  6.44k|            }
  157|  18.8k|            if (hdr->display_model_info_present)
  ------------------
  |  Branch (157:17): [True: 6.07k, False: 12.7k]
  ------------------
  158|  6.07k|                op->display_model_param_present = dav1d_get_bit(gb);
  159|  18.8k|            op->initial_display_delay =
  160|  18.8k|                op->display_model_param_present ? dav1d_get_bits(gb, 4) + 1 : 10;
  ------------------
  |  Branch (160:17): [True: 3.56k, False: 15.2k]
  ------------------
  161|  18.8k|        }
  162|       |#if DEBUG_SEQ_HDR
  163|       |        printf("SEQHDR: post-operating-points: off=%u\n",
  164|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  165|       |#endif
  166|  12.6k|    }
  167|       |
  168|  35.8k|    hdr->width_n_bits = dav1d_get_bits(gb, 4) + 1;
  169|  35.8k|    hdr->height_n_bits = dav1d_get_bits(gb, 4) + 1;
  170|  35.8k|    hdr->max_width = dav1d_get_bits(gb, hdr->width_n_bits) + 1;
  171|  35.8k|    hdr->max_height = dav1d_get_bits(gb, hdr->height_n_bits) + 1;
  172|       |#if DEBUG_SEQ_HDR
  173|       |    printf("SEQHDR: post-size: off=%u\n",
  174|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  175|       |#endif
  176|  35.8k|    if (!hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (176:9): [True: 12.0k, False: 23.7k]
  ------------------
  177|  12.0k|        hdr->frame_id_numbers_present = dav1d_get_bit(gb);
  178|  12.0k|        if (hdr->frame_id_numbers_present) {
  ------------------
  |  Branch (178:13): [True: 1.89k, False: 10.1k]
  ------------------
  179|  1.89k|            hdr->delta_frame_id_n_bits = dav1d_get_bits(gb, 4) + 2;
  180|  1.89k|            hdr->frame_id_n_bits = dav1d_get_bits(gb, 3) + hdr->delta_frame_id_n_bits + 1;
  181|  1.89k|        }
  182|  12.0k|    }
  183|       |#if DEBUG_SEQ_HDR
  184|       |    printf("SEQHDR: post-frame-id-numbers-present: off=%u\n",
  185|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  186|       |#endif
  187|       |
  188|  35.8k|    hdr->sb128 = dav1d_get_bit(gb);
  189|  35.8k|    hdr->filter_intra = dav1d_get_bit(gb);
  190|  35.8k|    hdr->intra_edge_filter = dav1d_get_bit(gb);
  191|  35.8k|    if (hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (191:9): [True: 23.7k, False: 12.0k]
  ------------------
  192|  23.7k|        hdr->screen_content_tools = DAV1D_ADAPTIVE;
  193|  23.7k|        hdr->force_integer_mv = DAV1D_ADAPTIVE;
  194|  23.7k|    } else {
  195|  12.0k|        hdr->inter_intra = dav1d_get_bit(gb);
  196|  12.0k|        hdr->masked_compound = dav1d_get_bit(gb);
  197|  12.0k|        hdr->warped_motion = dav1d_get_bit(gb);
  198|  12.0k|        hdr->dual_filter = dav1d_get_bit(gb);
  199|  12.0k|        hdr->order_hint = dav1d_get_bit(gb);
  200|  12.0k|        if (hdr->order_hint) {
  ------------------
  |  Branch (200:13): [True: 7.88k, False: 4.16k]
  ------------------
  201|  7.88k|            hdr->jnt_comp = dav1d_get_bit(gb);
  202|  7.88k|            hdr->ref_frame_mvs = dav1d_get_bit(gb);
  203|  7.88k|        }
  204|  12.0k|        hdr->screen_content_tools = dav1d_get_bit(gb) ? DAV1D_ADAPTIVE : dav1d_get_bit(gb);
  ------------------
  |  Branch (204:37): [True: 5.31k, False: 6.72k]
  ------------------
  205|       |    #if DEBUG_SEQ_HDR
  206|       |        printf("SEQHDR: post-screentools: off=%u\n",
  207|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  208|       |    #endif
  209|  12.0k|        hdr->force_integer_mv = hdr->screen_content_tools ?
  ------------------
  |  Branch (209:33): [True: 8.55k, False: 3.49k]
  ------------------
  210|  8.55k|                                dav1d_get_bit(gb) ? DAV1D_ADAPTIVE : dav1d_get_bit(gb) : 2;
  ------------------
  |  Branch (210:33): [True: 2.96k, False: 5.58k]
  ------------------
  211|  12.0k|        if (hdr->order_hint)
  ------------------
  |  Branch (211:13): [True: 7.88k, False: 4.16k]
  ------------------
  212|  7.88k|            hdr->order_hint_n_bits = dav1d_get_bits(gb, 3) + 1;
  213|  12.0k|    }
  214|  35.8k|    hdr->super_res = dav1d_get_bit(gb);
  215|  35.8k|    hdr->cdef = dav1d_get_bit(gb);
  216|  35.8k|    hdr->restoration = dav1d_get_bit(gb);
  217|       |#if DEBUG_SEQ_HDR
  218|       |    printf("SEQHDR: post-featurebits: off=%u\n",
  219|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  220|       |#endif
  221|       |
  222|  35.8k|    hdr->hbd = dav1d_get_bit(gb);
  223|  35.8k|    if (hdr->profile == 2 && hdr->hbd)
  ------------------
  |  Branch (223:9): [True: 13.2k, False: 22.5k]
  |  Branch (223:30): [True: 9.33k, False: 3.93k]
  ------------------
  224|  9.33k|        hdr->hbd += dav1d_get_bit(gb);
  225|  35.8k|    if (hdr->profile != 1)
  ------------------
  |  Branch (225:9): [True: 27.5k, False: 8.28k]
  ------------------
  226|  27.5k|        hdr->monochrome = dav1d_get_bit(gb);
  227|  35.8k|    hdr->color_description_present = dav1d_get_bit(gb);
  228|  35.8k|    if (hdr->color_description_present) {
  ------------------
  |  Branch (228:9): [True: 4.92k, False: 30.8k]
  ------------------
  229|  4.92k|        hdr->pri = dav1d_get_bits(gb, 8);
  230|  4.92k|        hdr->trc = dav1d_get_bits(gb, 8);
  231|  4.92k|        hdr->mtrx = dav1d_get_bits(gb, 8);
  232|  30.8k|    } else {
  233|  30.8k|        hdr->pri = DAV1D_COLOR_PRI_UNKNOWN;
  234|  30.8k|        hdr->trc = DAV1D_TRC_UNKNOWN;
  235|  30.8k|        hdr->mtrx = DAV1D_MC_UNKNOWN;
  236|  30.8k|    }
  237|  35.8k|    if (hdr->monochrome) {
  ------------------
  |  Branch (237:9): [True: 12.3k, False: 23.4k]
  ------------------
  238|  12.3k|        hdr->color_range = dav1d_get_bit(gb);
  239|  12.3k|        hdr->layout = DAV1D_PIXEL_LAYOUT_I400;
  240|  12.3k|        hdr->ss_hor = hdr->ss_ver = 1;
  241|  12.3k|        hdr->chr = DAV1D_CHR_UNKNOWN;
  242|  23.4k|    } else if (hdr->pri == DAV1D_COLOR_PRI_BT709 &&
  ------------------
  |  Branch (242:16): [True: 3.61k, False: 19.8k]
  ------------------
  243|  3.61k|               hdr->trc == DAV1D_TRC_SRGB &&
  ------------------
  |  Branch (243:16): [True: 2.14k, False: 1.47k]
  ------------------
  244|  2.14k|               hdr->mtrx == DAV1D_MC_IDENTITY)
  ------------------
  |  Branch (244:16): [True: 1.87k, False: 268]
  ------------------
  245|  1.87k|    {
  246|  1.87k|        hdr->layout = DAV1D_PIXEL_LAYOUT_I444;
  247|  1.87k|        hdr->color_range = 1;
  248|  1.87k|        if (hdr->profile != 1 && !(hdr->profile == 2 && hdr->hbd == 2))
  ------------------
  |  Branch (248:13): [True: 1.68k, False: 195]
  |  Branch (248:36): [True: 262, False: 1.41k]
  |  Branch (248:57): [True: 67, False: 195]
  ------------------
  249|  1.61k|            goto error;
  250|  21.5k|    } else {
  251|  21.5k|        hdr->color_range = dav1d_get_bit(gb);
  252|  21.5k|        switch (hdr->profile) {
  ------------------
  |  Branch (252:17): [True: 21.5k, False: 0]
  ------------------
  253|  7.55k|        case 0: hdr->layout = DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (253:9): [True: 7.55k, False: 13.9k]
  ------------------
  254|  7.55k|                hdr->ss_hor = hdr->ss_ver = 1;
  255|  7.55k|                break;
  256|  8.08k|        case 1: hdr->layout = DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (256:9): [True: 8.08k, False: 13.4k]
  ------------------
  257|  8.08k|                break;
  258|  5.90k|        case 2:
  ------------------
  |  Branch (258:9): [True: 5.90k, False: 15.6k]
  ------------------
  259|  5.90k|            if (hdr->hbd == 2) {
  ------------------
  |  Branch (259:17): [True: 3.12k, False: 2.78k]
  ------------------
  260|  3.12k|                hdr->ss_hor = dav1d_get_bit(gb);
  261|  3.12k|                if (hdr->ss_hor)
  ------------------
  |  Branch (261:21): [True: 1.21k, False: 1.90k]
  ------------------
  262|  1.21k|                    hdr->ss_ver = dav1d_get_bit(gb);
  263|  3.12k|            } else
  264|  2.78k|                hdr->ss_hor = 1;
  265|  5.90k|            hdr->layout = hdr->ss_hor ?
  ------------------
  |  Branch (265:27): [True: 4.00k, False: 1.90k]
  ------------------
  266|  4.00k|                          hdr->ss_ver ? DAV1D_PIXEL_LAYOUT_I420 :
  ------------------
  |  Branch (266:27): [True: 587, False: 3.41k]
  ------------------
  267|  4.00k|                                        DAV1D_PIXEL_LAYOUT_I422 :
  268|  5.90k|                                        DAV1D_PIXEL_LAYOUT_I444;
  269|  5.90k|            break;
  270|  21.5k|        }
  271|  21.5k|        hdr->chr = (hdr->ss_hor & hdr->ss_ver) ?
  ------------------
  |  Branch (271:20): [True: 8.14k, False: 13.4k]
  ------------------
  272|  13.4k|                   dav1d_get_bits(gb, 2) : DAV1D_CHR_UNKNOWN;
  273|  21.5k|    }
  274|  34.2k|    if (strict_std_compliance &&
  ------------------
  |  Branch (274:9): [True: 0, False: 34.2k]
  ------------------
  275|      0|        hdr->mtrx == DAV1D_MC_IDENTITY && hdr->layout != DAV1D_PIXEL_LAYOUT_I444)
  ------------------
  |  Branch (275:9): [True: 0, False: 0]
  |  Branch (275:43): [True: 0, False: 0]
  ------------------
  276|      0|    {
  277|      0|        goto error;
  278|      0|    }
  279|  34.2k|    if (!hdr->monochrome)
  ------------------
  |  Branch (279:9): [True: 21.8k, False: 12.3k]
  ------------------
  280|  21.8k|        hdr->separate_uv_delta_q = dav1d_get_bit(gb);
  281|       |#if DEBUG_SEQ_HDR
  282|       |    printf("SEQHDR: post-colorinfo: off=%u\n",
  283|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  284|       |#endif
  285|       |
  286|  34.2k|    hdr->film_grain_present = dav1d_get_bit(gb);
  287|       |#if DEBUG_SEQ_HDR
  288|       |    printf("SEQHDR: post-filmgrain: off=%u\n",
  289|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  290|       |#endif
  291|       |
  292|       |    // We needn't bother flushing the OBU here: we'll check we didn't
  293|       |    // overrun in the caller and will then discard gb, so there's no
  294|       |    // point in setting its position properly.
  295|       |
  296|  34.2k|    return check_trailing_bits(gb, strict_std_compliance);
  297|       |
  298|  3.29k|error:
  299|  3.29k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  3.29k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  300|  34.2k|}
obu.c:parse_frame_hdr:
  409|  76.0k|static int parse_frame_hdr(Dav1dContext *const c, GetBits *const gb) {
  410|  76.0k|#define DEBUG_FRAME_HDR 0
  411|       |
  412|       |#if DEBUG_FRAME_HDR
  413|       |    const uint8_t *const init_ptr = gb->ptr;
  414|       |#endif
  415|  76.0k|    const Dav1dSequenceHeader *const seqhdr = c->seq_hdr;
  416|  76.0k|    Dav1dFrameHeader *const hdr = c->frame_hdr;
  417|       |
  418|  76.0k|    if (!seqhdr->reduced_still_picture_header)
  ------------------
  |  Branch (418:9): [True: 41.0k, False: 35.0k]
  ------------------
  419|  41.0k|        hdr->show_existing_frame = dav1d_get_bit(gb);
  420|       |#if DEBUG_FRAME_HDR
  421|       |    printf("HDR: post-show_existing_frame: off=%td\n",
  422|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  423|       |#endif
  424|  76.0k|    if (hdr->show_existing_frame) {
  ------------------
  |  Branch (424:9): [True: 5.66k, False: 70.3k]
  ------------------
  425|  5.66k|        hdr->existing_frame_idx = dav1d_get_bits(gb, 3);
  426|  5.66k|        if (seqhdr->decoder_model_info_present && !seqhdr->equal_picture_interval)
  ------------------
  |  Branch (426:13): [True: 549, False: 5.11k]
  |  Branch (426:51): [True: 203, False: 346]
  ------------------
  427|    203|            hdr->frame_presentation_delay = dav1d_get_bits(gb, seqhdr->frame_presentation_delay_length);
  428|  5.66k|        if (seqhdr->frame_id_numbers_present) {
  ------------------
  |  Branch (428:13): [True: 858, False: 4.80k]
  ------------------
  429|    858|            hdr->frame_id = dav1d_get_bits(gb, seqhdr->frame_id_n_bits);
  430|    858|            Dav1dFrameHeader *const ref_frame_hdr = c->refs[hdr->existing_frame_idx].p.p.frame_hdr;
  431|    858|            if (!ref_frame_hdr || ref_frame_hdr->frame_id != hdr->frame_id) goto error;
  ------------------
  |  Branch (431:17): [True: 229, False: 629]
  |  Branch (431:35): [True: 199, False: 430]
  ------------------
  432|    858|        }
  433|  5.23k|        return 0;
  434|  5.66k|    }
  435|       |
  436|  70.3k|    if (seqhdr->reduced_still_picture_header) {
  ------------------
  |  Branch (436:9): [True: 35.0k, False: 35.3k]
  ------------------
  437|  35.0k|        hdr->frame_type = DAV1D_FRAME_TYPE_KEY;
  438|  35.0k|        hdr->show_frame = 1;
  439|  35.3k|    } else {
  440|  35.3k|        hdr->frame_type = dav1d_get_bits(gb, 2);
  441|  35.3k|        hdr->show_frame = dav1d_get_bit(gb);
  442|  35.3k|    }
  443|  70.3k|    if (hdr->show_frame) {
  ------------------
  |  Branch (443:9): [True: 61.8k, False: 8.50k]
  ------------------
  444|  61.8k|        if (seqhdr->decoder_model_info_present && !seqhdr->equal_picture_interval)
  ------------------
  |  Branch (444:13): [True: 1.54k, False: 60.3k]
  |  Branch (444:51): [True: 1.27k, False: 273]
  ------------------
  445|  1.27k|            hdr->frame_presentation_delay = dav1d_get_bits(gb, seqhdr->frame_presentation_delay_length);
  446|  61.8k|        hdr->showable_frame = hdr->frame_type != DAV1D_FRAME_TYPE_KEY;
  447|  61.8k|    } else
  448|  8.50k|        hdr->showable_frame = dav1d_get_bit(gb);
  449|  70.3k|    hdr->error_resilient_mode =
  450|  70.3k|        (hdr->frame_type == DAV1D_FRAME_TYPE_KEY && hdr->show_frame) ||
  ------------------
  |  Branch (450:10): [True: 44.4k, False: 25.9k]
  |  Branch (450:53): [True: 42.9k, False: 1.52k]
  ------------------
  451|  27.4k|        hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ||
  ------------------
  |  Branch (451:9): [True: 2.87k, False: 24.5k]
  ------------------
  452|  24.5k|        seqhdr->reduced_still_picture_header || dav1d_get_bit(gb);
  ------------------
  |  Branch (452:9): [True: 0, False: 24.5k]
  |  Branch (452:49): [True: 1.51k, False: 23.0k]
  ------------------
  453|       |#if DEBUG_FRAME_HDR
  454|       |    printf("HDR: post-frametype_bits: off=%td\n",
  455|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  456|       |#endif
  457|  70.3k|    hdr->disable_cdf_update = dav1d_get_bit(gb);
  458|  70.3k|    hdr->allow_screen_content_tools = seqhdr->screen_content_tools == DAV1D_ADAPTIVE ?
  ------------------
  |  Branch (458:39): [True: 49.1k, False: 21.2k]
  ------------------
  459|  49.1k|                                      dav1d_get_bit(gb) : seqhdr->screen_content_tools;
  460|  70.3k|    if (hdr->allow_screen_content_tools)
  ------------------
  |  Branch (460:9): [True: 43.2k, False: 27.1k]
  ------------------
  461|  43.2k|        hdr->force_integer_mv = seqhdr->force_integer_mv == DAV1D_ADAPTIVE ?
  ------------------
  |  Branch (461:33): [True: 27.4k, False: 15.8k]
  ------------------
  462|  27.4k|                                dav1d_get_bit(gb) : seqhdr->force_integer_mv;
  463|       |
  464|  70.3k|    if (IS_KEY_OR_INTRA(hdr))
  ------------------
  |  |   43|  70.3k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  70.3k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 45.5k, False: 24.8k]
  |  |  ------------------
  ------------------
  465|  45.5k|        hdr->force_integer_mv = 1;
  466|       |
  467|  70.3k|    if (seqhdr->frame_id_numbers_present)
  ------------------
  |  Branch (467:9): [True: 1.10k, False: 69.2k]
  ------------------
  468|  1.10k|        hdr->frame_id = dav1d_get_bits(gb, seqhdr->frame_id_n_bits);
  469|       |
  470|  70.3k|    if (!seqhdr->reduced_still_picture_header)
  ------------------
  |  Branch (470:9): [True: 35.3k, False: 35.0k]
  ------------------
  471|  35.3k|        hdr->frame_size_override = hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ? 1 : dav1d_get_bit(gb);
  ------------------
  |  Branch (471:36): [True: 2.87k, False: 32.4k]
  ------------------
  472|       |#if DEBUG_FRAME_HDR
  473|       |    printf("HDR: post-frame_size_override_flag: off=%td\n",
  474|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  475|       |#endif
  476|  70.3k|    if (seqhdr->order_hint)
  ------------------
  |  Branch (476:9): [True: 23.7k, False: 46.6k]
  ------------------
  477|  23.7k|        hdr->frame_offset = dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  478|  70.3k|    hdr->primary_ref_frame = !hdr->error_resilient_mode && IS_INTER_OR_SWITCH(hdr) ?
  ------------------
  |  |   36|  23.0k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 21.3k, False: 1.69k]
  |  |  ------------------
  ------------------
  |  Branch (478:30): [True: 23.0k, False: 47.3k]
  ------------------
  479|  49.0k|                             dav1d_get_bits(gb, 3) : DAV1D_PRIMARY_REF_NONE;
  ------------------
  |  |   45|   119k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  480|       |
  481|  70.3k|    if (seqhdr->decoder_model_info_present) {
  ------------------
  |  Branch (481:9): [True: 1.57k, False: 68.8k]
  ------------------
  482|  1.57k|        hdr->buffer_removal_time_present = dav1d_get_bit(gb);
  483|  1.57k|        if (hdr->buffer_removal_time_present) {
  ------------------
  |  Branch (483:13): [True: 861, False: 714]
  ------------------
  484|  6.12k|            for (int i = 0; i < c->seq_hdr->num_operating_points; i++) {
  ------------------
  |  Branch (484:29): [True: 5.26k, False: 861]
  ------------------
  485|  5.26k|                const struct Dav1dSequenceHeaderOperatingPoint *const seqop = &seqhdr->operating_points[i];
  486|  5.26k|                struct Dav1dFrameHeaderOperatingPoint *const op = &hdr->operating_points[i];
  487|  5.26k|                if (seqop->decoder_model_param_present) {
  ------------------
  |  Branch (487:21): [True: 4.36k, False: 906]
  ------------------
  488|  4.36k|                    int in_temporal_layer = (seqop->idc >> hdr->temporal_id) & 1;
  489|  4.36k|                    int in_spatial_layer  = (seqop->idc >> (hdr->spatial_id + 8)) & 1;
  490|  4.36k|                    if (!seqop->idc || (in_temporal_layer && in_spatial_layer))
  ------------------
  |  Branch (490:25): [True: 293, False: 4.06k]
  |  Branch (490:41): [True: 3.23k, False: 832]
  |  Branch (490:62): [True: 2.40k, False: 829]
  ------------------
  491|  2.70k|                        op->buffer_removal_time = dav1d_get_bits(gb, seqhdr->buffer_removal_delay_length);
  492|  4.36k|                }
  493|  5.26k|            }
  494|    861|        }
  495|  1.57k|    }
  496|       |
  497|  70.3k|    if (IS_KEY_OR_INTRA(hdr)) {
  ------------------
  |  |   43|  70.3k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  70.3k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 45.5k, False: 24.8k]
  |  |  ------------------
  ------------------
  498|  45.5k|        hdr->refresh_frame_flags = (hdr->frame_type == DAV1D_FRAME_TYPE_KEY &&
  ------------------
  |  Branch (498:37): [True: 44.4k, False: 1.08k]
  ------------------
  499|  44.4k|                                    hdr->show_frame) ? 0xff : dav1d_get_bits(gb, 8);
  ------------------
  |  Branch (499:37): [True: 42.9k, False: 1.52k]
  ------------------
  500|  45.5k|        if (hdr->refresh_frame_flags != 0xff && hdr->error_resilient_mode && seqhdr->order_hint)
  ------------------
  |  Branch (500:13): [True: 2.52k, False: 43.0k]
  |  Branch (500:49): [True: 871, False: 1.65k]
  |  Branch (500:78): [True: 561, False: 310]
  ------------------
  501|  5.04k|            for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (501:29): [True: 4.48k, False: 561]
  ------------------
  502|  4.48k|                dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  503|  45.5k|        if (c->strict_std_compliance &&
  ------------------
  |  Branch (503:13): [True: 0, False: 45.5k]
  ------------------
  504|      0|            hdr->frame_type == DAV1D_FRAME_TYPE_INTRA && hdr->refresh_frame_flags == 0xff)
  ------------------
  |  Branch (504:13): [True: 0, False: 0]
  |  Branch (504:58): [True: 0, False: 0]
  ------------------
  505|      0|        {
  506|      0|            goto error;
  507|      0|        }
  508|  45.5k|        if (read_frame_size(c, gb, 0) < 0) goto error;
  ------------------
  |  Branch (508:13): [True: 0, False: 45.5k]
  ------------------
  509|  45.5k|        if (hdr->allow_screen_content_tools && !hdr->super_res.enabled)
  ------------------
  |  Branch (509:13): [True: 27.3k, False: 18.1k]
  |  Branch (509:48): [True: 26.2k, False: 1.16k]
  ------------------
  510|  26.2k|            hdr->allow_intrabc = dav1d_get_bit(gb);
  511|  45.5k|    } else {
  512|  24.8k|        hdr->refresh_frame_flags = hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ? 0xff :
  ------------------
  |  Branch (512:36): [True: 2.87k, False: 21.9k]
  ------------------
  513|  24.8k|                                   dav1d_get_bits(gb, 8);
  514|  24.8k|        if (hdr->error_resilient_mode && seqhdr->order_hint)
  ------------------
  |  Branch (514:13): [True: 3.47k, False: 21.3k]
  |  Branch (514:42): [True: 2.15k, False: 1.32k]
  ------------------
  515|  19.4k|            for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (515:29): [True: 17.2k, False: 2.15k]
  ------------------
  516|  17.2k|                dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  517|  24.8k|        if (seqhdr->order_hint) {
  ------------------
  |  Branch (517:13): [True: 19.1k, False: 5.72k]
  ------------------
  518|  19.1k|            hdr->frame_ref_short_signaling = dav1d_get_bit(gb);
  519|  19.1k|            if (hdr->frame_ref_short_signaling) {
  ------------------
  |  Branch (519:17): [True: 14.3k, False: 4.78k]
  ------------------
  520|  14.3k|                hdr->refidx[0] = dav1d_get_bits(gb, 3);
  521|  14.3k|                hdr->refidx[1] = hdr->refidx[2] = -1;
  522|  14.3k|                hdr->refidx[3] = dav1d_get_bits(gb, 3);
  523|       |
  524|       |                /* +1 allows for unconditional stores, as unused
  525|       |                 * values can be dumped into frame_offset[-1]. */
  526|  14.3k|                int frame_offset_mem[8+1];
  527|  14.3k|                int *const frame_offset = &frame_offset_mem[1];
  528|  14.3k|                int earliest_ref = -1;
  529|   126k|                for (int i = 0, earliest_offset = INT_MAX; i < 8; i++) {
  ------------------
  |  Branch (529:60): [True: 112k, False: 13.7k]
  ------------------
  530|   112k|                    const Dav1dFrameHeader *const refhdr = c->refs[i].p.p.frame_hdr;
  531|   112k|                    if (!refhdr) goto error;
  ------------------
  |  Branch (531:25): [True: 580, False: 112k]
  ------------------
  532|   112k|                    const int diff = get_poc_diff(seqhdr->order_hint_n_bits,
  533|   112k|                                                  refhdr->frame_offset,
  534|   112k|                                                  hdr->frame_offset);
  535|   112k|                    frame_offset[i] = diff;
  536|   112k|                    if (diff < earliest_offset) {
  ------------------
  |  Branch (536:25): [True: 18.0k, False: 94.3k]
  ------------------
  537|  18.0k|                        earliest_offset = diff;
  538|  18.0k|                        earliest_ref = i;
  539|  18.0k|                    }
  540|   112k|                }
  541|  13.7k|                frame_offset[hdr->refidx[0]] = INT_MIN; // = reference frame is used
  542|  13.7k|                frame_offset[hdr->refidx[3]] = INT_MIN;
  543|  13.7k|                assert(earliest_ref >= 0);
  ------------------
  |  Branch (543:17): [True: 13.7k, False: 0]
  ------------------
  544|       |
  545|  13.7k|                int refidx = -1;
  546|   123k|                for (int i = 0, latest_offset = 0; i < 8; i++) {
  ------------------
  |  Branch (546:52): [True: 109k, False: 13.7k]
  ------------------
  547|   109k|                    const int hint = frame_offset[i];
  548|   109k|                    if (hint >= latest_offset) {
  ------------------
  |  Branch (548:25): [True: 61.4k, False: 48.4k]
  ------------------
  549|  61.4k|                        latest_offset = hint;
  550|  61.4k|                        refidx = i;
  551|  61.4k|                    }
  552|   109k|                }
  553|  13.7k|                frame_offset[refidx] = INT_MIN;
  554|  13.7k|                hdr->refidx[6] = refidx;
  555|       |
  556|  41.2k|                for (int i = 4; i < 6; i++) {
  ------------------
  |  Branch (556:33): [True: 27.4k, False: 13.7k]
  ------------------
  557|       |                    /* Unsigned compares to handle negative values. */
  558|  27.4k|                    unsigned earliest_offset = UINT8_MAX;
  559|  27.4k|                    refidx = -1;
  560|   247k|                    for (int j = 0; j < 8; j++) {
  ------------------
  |  Branch (560:37): [True: 219k, False: 27.4k]
  ------------------
  561|   219k|                        const unsigned hint = frame_offset[j];
  562|   219k|                        if (hint < earliest_offset) {
  ------------------
  |  Branch (562:29): [True: 23.7k, False: 196k]
  ------------------
  563|  23.7k|                            earliest_offset = hint;
  564|  23.7k|                            refidx = j;
  565|  23.7k|                        }
  566|   219k|                    }
  567|  27.4k|                    frame_offset[refidx] = INT_MIN;
  568|  27.4k|                    hdr->refidx[i] = refidx;
  569|  27.4k|                }
  570|       |
  571|  96.1k|                for (int i = 1; i < 7; i++) {
  ------------------
  |  Branch (571:33): [True: 82.4k, False: 13.7k]
  ------------------
  572|  82.4k|                    refidx = hdr->refidx[i];
  573|  82.4k|                    if (refidx < 0) {
  ------------------
  |  Branch (573:25): [True: 33.7k, False: 48.6k]
  ------------------
  574|  33.7k|                        unsigned latest_offset = ~UINT8_MAX;
  575|   303k|                        for (int j = 0; j < 8; j++) {
  ------------------
  |  Branch (575:41): [True: 269k, False: 33.7k]
  ------------------
  576|   269k|                            const unsigned hint = frame_offset[j];
  577|   269k|                            if (hint >= latest_offset) {
  ------------------
  |  Branch (577:33): [True: 47.1k, False: 222k]
  ------------------
  578|  47.1k|                                latest_offset = hint;
  579|  47.1k|                                refidx = j;
  580|  47.1k|                            }
  581|   269k|                        }
  582|  33.7k|                        frame_offset[refidx] = INT_MIN;
  583|  33.7k|                        hdr->refidx[i] = refidx >= 0 ? refidx : earliest_ref;
  ------------------
  |  Branch (583:42): [True: 14.8k, False: 18.9k]
  ------------------
  584|  33.7k|                    }
  585|  82.4k|                }
  586|  13.7k|            }
  587|  19.1k|        }
  588|   189k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (588:25): [True: 165k, False: 23.6k]
  ------------------
  589|   165k|            if (!hdr->frame_ref_short_signaling)
  ------------------
  |  Branch (589:17): [True: 70.8k, False: 95.1k]
  ------------------
  590|  70.8k|                hdr->refidx[i] = dav1d_get_bits(gb, 3);
  591|   165k|            if (seqhdr->frame_id_numbers_present) {
  ------------------
  |  Branch (591:17): [True: 848, False: 165k]
  ------------------
  592|    848|                const unsigned delta_ref_frame_id = dav1d_get_bits(gb, seqhdr->delta_frame_id_n_bits) + 1;
  593|    848|                const unsigned ref_frame_id = (hdr->frame_id + (1 << seqhdr->frame_id_n_bits) - delta_ref_frame_id) & ((1 << seqhdr->frame_id_n_bits) - 1);
  594|    848|                Dav1dFrameHeader *const ref_frame_hdr = c->refs[hdr->refidx[i]].p.p.frame_hdr;
  595|    848|                if (!ref_frame_hdr || ref_frame_hdr->frame_id != ref_frame_id) goto error;
  ------------------
  |  Branch (595:21): [True: 356, False: 492]
  |  Branch (595:39): [True: 276, False: 216]
  ------------------
  596|    848|            }
  597|   165k|        }
  598|  23.6k|        const int use_ref = !hdr->error_resilient_mode &&
  ------------------
  |  Branch (598:29): [True: 20.8k, False: 2.76k]
  ------------------
  599|  20.8k|                            hdr->frame_size_override;
  ------------------
  |  Branch (599:29): [True: 12.3k, False: 8.47k]
  ------------------
  600|  23.6k|        if (read_frame_size(c, gb, use_ref) < 0) goto error;
  ------------------
  |  Branch (600:13): [True: 242, False: 23.3k]
  ------------------
  601|  23.3k|        if (!hdr->force_integer_mv)
  ------------------
  |  Branch (601:13): [True: 12.7k, False: 10.5k]
  ------------------
  602|  12.7k|            hdr->hp = dav1d_get_bit(gb);
  603|  23.3k|        hdr->subpel_filter_mode = dav1d_get_bit(gb) ? DAV1D_FILTER_SWITCHABLE :
  ------------------
  |  Branch (603:35): [True: 3.60k, False: 19.7k]
  ------------------
  604|  23.3k|                                                      dav1d_get_bits(gb, 2);
  605|  23.3k|        hdr->switchable_motion_mode = dav1d_get_bit(gb);
  606|  23.3k|        if (!hdr->error_resilient_mode && seqhdr->ref_frame_mvs &&
  ------------------
  |  Branch (606:13): [True: 20.6k, False: 2.76k]
  |  Branch (606:43): [True: 14.5k, False: 6.06k]
  ------------------
  607|  14.5k|            seqhdr->order_hint && IS_INTER_OR_SWITCH(hdr))
  ------------------
  |  |   36|  14.5k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 14.5k, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (607:13): [True: 14.5k, False: 0]
  ------------------
  608|  14.5k|        {
  609|  14.5k|            hdr->use_ref_frame_mvs = dav1d_get_bit(gb);
  610|  14.5k|        }
  611|  23.3k|    }
  612|       |#if DEBUG_FRAME_HDR
  613|       |    printf("HDR: post-frametype-specific-bits: off=%td\n",
  614|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  615|       |#endif
  616|       |
  617|  68.9k|    if (!seqhdr->reduced_still_picture_header && !hdr->disable_cdf_update)
  ------------------
  |  Branch (617:9): [True: 33.8k, False: 35.0k]
  |  Branch (617:50): [True: 27.6k, False: 6.27k]
  ------------------
  618|  27.6k|        hdr->refresh_context = !dav1d_get_bit(gb);
  619|       |#if DEBUG_FRAME_HDR
  620|       |    printf("HDR: post-refresh_context: off=%td\n",
  621|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  622|       |#endif
  623|       |
  624|       |    // tile data
  625|  68.9k|    hdr->tiling.uniform = dav1d_get_bit(gb);
  626|  68.9k|    const int sbsz_min1 = (64 << seqhdr->sb128) - 1;
  627|  68.9k|    const int sbsz_log2 = 6 + seqhdr->sb128;
  628|  68.9k|    const int sbw = (hdr->width[0] + sbsz_min1) >> sbsz_log2;
  629|  68.9k|    const int sbh = (hdr->height + sbsz_min1) >> sbsz_log2;
  630|  68.9k|    const int max_tile_width_sb = 4096 >> sbsz_log2;
  631|  68.9k|    const int max_tile_area_sb = 4096 * 2304 >> (2 * sbsz_log2);
  632|  68.9k|    hdr->tiling.min_log2_cols = tile_log2(max_tile_width_sb, sbw);
  633|  68.9k|    hdr->tiling.max_log2_cols = tile_log2(1, imin(sbw, DAV1D_MAX_TILE_COLS));
  ------------------
  |  |   41|  68.9k|#define DAV1D_MAX_TILE_COLS 64
  ------------------
  634|  68.9k|    hdr->tiling.max_log2_rows = tile_log2(1, imin(sbh, DAV1D_MAX_TILE_ROWS));
  ------------------
  |  |   42|  68.9k|#define DAV1D_MAX_TILE_ROWS 64
  ------------------
  635|  68.9k|    const int min_log2_tiles = imax(tile_log2(max_tile_area_sb, sbw * sbh),
  636|  68.9k|                              hdr->tiling.min_log2_cols);
  637|  68.9k|    if (hdr->tiling.uniform) {
  ------------------
  |  Branch (637:9): [True: 42.6k, False: 26.3k]
  ------------------
  638|  42.6k|        for (hdr->tiling.log2_cols = hdr->tiling.min_log2_cols;
  639|  44.3k|             hdr->tiling.log2_cols < hdr->tiling.max_log2_cols && dav1d_get_bit(gb);
  ------------------
  |  Branch (639:14): [True: 18.9k, False: 25.3k]
  |  Branch (639:67): [True: 1.70k, False: 17.2k]
  ------------------
  640|  42.6k|             hdr->tiling.log2_cols++) ;
  641|  42.6k|        const int tile_w = 1 + ((sbw - 1) >> hdr->tiling.log2_cols);
  642|  42.6k|        hdr->tiling.cols = 0;
  643|  98.7k|        for (int sbx = 0; sbx < sbw; sbx += tile_w, hdr->tiling.cols++)
  ------------------
  |  Branch (643:27): [True: 56.1k, False: 42.6k]
  ------------------
  644|  56.1k|            hdr->tiling.col_start_sb[hdr->tiling.cols] = sbx;
  645|  42.6k|        hdr->tiling.min_log2_rows =
  646|  42.6k|            imax(min_log2_tiles - hdr->tiling.log2_cols, 0);
  647|       |
  648|  42.6k|        for (hdr->tiling.log2_rows = hdr->tiling.min_log2_rows;
  649|  43.8k|             hdr->tiling.log2_rows < hdr->tiling.max_log2_rows && dav1d_get_bit(gb);
  ------------------
  |  Branch (649:14): [True: 9.03k, False: 34.8k]
  |  Branch (649:67): [True: 1.25k, False: 7.77k]
  ------------------
  650|  42.6k|             hdr->tiling.log2_rows++) ;
  651|  42.6k|        const int tile_h = 1 + ((sbh - 1) >> hdr->tiling.log2_rows);
  652|  42.6k|        hdr->tiling.rows = 0;
  653|  95.6k|        for (int sby = 0; sby < sbh; sby += tile_h, hdr->tiling.rows++)
  ------------------
  |  Branch (653:27): [True: 53.0k, False: 42.6k]
  ------------------
  654|  53.0k|            hdr->tiling.row_start_sb[hdr->tiling.rows] = sby;
  655|  42.6k|    } else {
  656|  26.3k|        hdr->tiling.cols = 0;
  657|  26.3k|        int widest_tile = 0, max_tile_area_sb = sbw * sbh;
  658|   108k|        for (int sbx = 0; sbx < sbw && hdr->tiling.cols < DAV1D_MAX_TILE_COLS; hdr->tiling.cols++) {
  ------------------
  |  |   41|  82.7k|#define DAV1D_MAX_TILE_COLS 64
  ------------------
  |  Branch (658:27): [True: 82.7k, False: 25.6k]
  |  Branch (658:40): [True: 82.0k, False: 652]
  ------------------
  659|  82.0k|            const int tile_width_sb = imin(sbw - sbx, max_tile_width_sb);
  660|  82.0k|            const int tile_w = (tile_width_sb > 1) ? 1 + dav1d_get_uniform(gb, tile_width_sb) : 1;
  ------------------
  |  Branch (660:32): [True: 57.9k, False: 24.1k]
  ------------------
  661|  82.0k|            hdr->tiling.col_start_sb[hdr->tiling.cols] = sbx;
  662|  82.0k|            sbx += tile_w;
  663|  82.0k|            widest_tile = imax(widest_tile, tile_w);
  664|  82.0k|        }
  665|  26.3k|        hdr->tiling.log2_cols = tile_log2(1, hdr->tiling.cols);
  666|  26.3k|        if (min_log2_tiles) max_tile_area_sb >>= min_log2_tiles + 1;
  ------------------
  |  Branch (666:13): [True: 776, False: 25.5k]
  ------------------
  667|  26.3k|        const int max_tile_height_sb = imax(max_tile_area_sb / widest_tile, 1);
  668|       |
  669|  26.3k|        hdr->tiling.rows = 0;
  670|   121k|        for (int sby = 0; sby < sbh && hdr->tiling.rows < DAV1D_MAX_TILE_ROWS; hdr->tiling.rows++) {
  ------------------
  |  |   42|  96.4k|#define DAV1D_MAX_TILE_ROWS 64
  ------------------
  |  Branch (670:27): [True: 96.4k, False: 25.4k]
  |  Branch (670:40): [True: 95.5k, False: 843]
  ------------------
  671|  95.5k|            const int tile_height_sb = imin(sbh - sby, max_tile_height_sb);
  672|  95.5k|            const int tile_h = (tile_height_sb > 1) ? 1 + dav1d_get_uniform(gb, tile_height_sb) : 1;
  ------------------
  |  Branch (672:32): [True: 70.4k, False: 25.1k]
  ------------------
  673|  95.5k|            hdr->tiling.row_start_sb[hdr->tiling.rows] = sby;
  674|  95.5k|            sby += tile_h;
  675|  95.5k|        }
  676|  26.3k|        hdr->tiling.log2_rows = tile_log2(1, hdr->tiling.rows);
  677|  26.3k|    }
  678|  68.9k|    hdr->tiling.col_start_sb[hdr->tiling.cols] = sbw;
  679|  68.9k|    hdr->tiling.row_start_sb[hdr->tiling.rows] = sbh;
  680|  68.9k|    if (hdr->tiling.log2_cols || hdr->tiling.log2_rows) {
  ------------------
  |  Branch (680:9): [True: 5.93k, False: 63.0k]
  |  Branch (680:34): [True: 2.01k, False: 60.9k]
  ------------------
  681|  7.94k|        hdr->tiling.update = dav1d_get_bits(gb, hdr->tiling.log2_cols + hdr->tiling.log2_rows);
  682|  7.94k|        if (hdr->tiling.update >= hdr->tiling.cols * hdr->tiling.rows)
  ------------------
  |  Branch (682:13): [True: 236, False: 7.70k]
  ------------------
  683|    236|            goto error;
  684|  7.70k|        hdr->tiling.n_bytes = dav1d_get_bits(gb, 2) + 1;
  685|  7.70k|    }
  686|       |#if DEBUG_FRAME_HDR
  687|       |    printf("HDR: post-tiling: off=%td\n",
  688|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  689|       |#endif
  690|       |
  691|       |    // quant data
  692|  68.6k|    hdr->quant.yac = dav1d_get_bits(gb, 8);
  693|  68.6k|    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (693:9): [True: 8.03k, False: 60.6k]
  ------------------
  694|  8.03k|        hdr->quant.ydc_delta = dav1d_get_sbits(gb, 7);
  695|  68.6k|    if (!seqhdr->monochrome) {
  ------------------
  |  Branch (695:9): [True: 40.4k, False: 28.2k]
  ------------------
  696|       |        // If the sequence header says that delta_q might be different
  697|       |        // for U, V, we must check whether it actually is for this
  698|       |        // frame.
  699|  40.4k|        const int diff_uv_delta = seqhdr->separate_uv_delta_q ? dav1d_get_bit(gb) : 0;
  ------------------
  |  Branch (699:35): [True: 12.3k, False: 28.1k]
  ------------------
  700|  40.4k|        if (dav1d_get_bit(gb))
  ------------------
  |  Branch (700:13): [True: 3.71k, False: 36.7k]
  ------------------
  701|  3.71k|            hdr->quant.udc_delta = dav1d_get_sbits(gb, 7);
  702|  40.4k|        if (dav1d_get_bit(gb))
  ------------------
  |  Branch (702:13): [True: 4.19k, False: 36.2k]
  ------------------
  703|  4.19k|            hdr->quant.uac_delta = dav1d_get_sbits(gb, 7);
  704|  40.4k|        if (diff_uv_delta) {
  ------------------
  |  Branch (704:13): [True: 3.56k, False: 36.9k]
  ------------------
  705|  3.56k|            if (dav1d_get_bit(gb))
  ------------------
  |  Branch (705:17): [True: 2.61k, False: 952]
  ------------------
  706|  2.61k|                hdr->quant.vdc_delta = dav1d_get_sbits(gb, 7);
  707|  3.56k|            if (dav1d_get_bit(gb))
  ------------------
  |  Branch (707:17): [True: 1.25k, False: 2.31k]
  ------------------
  708|  1.25k|                hdr->quant.vac_delta = dav1d_get_sbits(gb, 7);
  709|  36.9k|        } else {
  710|  36.9k|            hdr->quant.vdc_delta = hdr->quant.udc_delta;
  711|  36.9k|            hdr->quant.vac_delta = hdr->quant.uac_delta;
  712|  36.9k|        }
  713|  40.4k|    }
  714|       |#if DEBUG_FRAME_HDR
  715|       |    printf("HDR: post-quant: off=%td\n",
  716|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  717|       |#endif
  718|  68.6k|    hdr->quant.qm = dav1d_get_bit(gb);
  719|  68.6k|    if (hdr->quant.qm) {
  ------------------
  |  Branch (719:9): [True: 8.62k, False: 60.0k]
  ------------------
  720|  8.62k|        hdr->quant.qm_y = dav1d_get_bits(gb, 4);
  721|  8.62k|        hdr->quant.qm_u = dav1d_get_bits(gb, 4);
  722|  8.62k|        hdr->quant.qm_v = seqhdr->separate_uv_delta_q ? dav1d_get_bits(gb, 4) :
  ------------------
  |  Branch (722:27): [True: 1.64k, False: 6.98k]
  ------------------
  723|  8.62k|                                                        hdr->quant.qm_u;
  724|  8.62k|    }
  725|       |#if DEBUG_FRAME_HDR
  726|       |    printf("HDR: post-qm: off=%td\n",
  727|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  728|       |#endif
  729|       |
  730|       |    // segmentation data
  731|  68.6k|    hdr->segmentation.enabled = dav1d_get_bit(gb);
  732|  68.6k|    if (hdr->segmentation.enabled) {
  ------------------
  |  Branch (732:9): [True: 13.0k, False: 55.6k]
  ------------------
  733|  13.0k|        if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  13.0k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (733:13): [True: 3.20k, False: 9.83k]
  ------------------
  734|  3.20k|            hdr->segmentation.update_map = 1;
  735|  3.20k|            hdr->segmentation.update_data = 1;
  736|  9.83k|        } else {
  737|  9.83k|            hdr->segmentation.update_map = dav1d_get_bit(gb);
  738|  9.83k|            if (hdr->segmentation.update_map)
  ------------------
  |  Branch (738:17): [True: 1.58k, False: 8.24k]
  ------------------
  739|  1.58k|                hdr->segmentation.temporal = dav1d_get_bit(gb);
  740|  9.83k|            hdr->segmentation.update_data = dav1d_get_bit(gb);
  741|  9.83k|        }
  742|       |
  743|  13.0k|        if (hdr->segmentation.update_data) {
  ------------------
  |  Branch (743:13): [True: 4.87k, False: 8.16k]
  ------------------
  744|  4.87k|            hdr->segmentation.seg_data.last_active_segid = -1;
  745|  43.8k|            for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++) {
  ------------------
  |  |   43|  43.8k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (745:29): [True: 39.0k, False: 4.87k]
  ------------------
  746|  39.0k|                Dav1dSegmentationData *const seg =
  747|  39.0k|                    &hdr->segmentation.seg_data.d[i];
  748|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (748:21): [True: 6.39k, False: 32.6k]
  ------------------
  749|  6.39k|                    seg->delta_q = dav1d_get_sbits(gb, 9);
  750|  6.39k|                    hdr->segmentation.seg_data.last_active_segid = i;
  751|  6.39k|                }
  752|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (752:21): [True: 4.84k, False: 34.1k]
  ------------------
  753|  4.84k|                    seg->delta_lf_y_v = dav1d_get_sbits(gb, 7);
  754|  4.84k|                    hdr->segmentation.seg_data.last_active_segid = i;
  755|  4.84k|                }
  756|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (756:21): [True: 5.60k, False: 33.4k]
  ------------------
  757|  5.60k|                    seg->delta_lf_y_h = dav1d_get_sbits(gb, 7);
  758|  5.60k|                    hdr->segmentation.seg_data.last_active_segid = i;
  759|  5.60k|                }
  760|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (760:21): [True: 4.65k, False: 34.3k]
  ------------------
  761|  4.65k|                    seg->delta_lf_u = dav1d_get_sbits(gb, 7);
  762|  4.65k|                    hdr->segmentation.seg_data.last_active_segid = i;
  763|  4.65k|                }
  764|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (764:21): [True: 3.55k, False: 35.4k]
  ------------------
  765|  3.55k|                    seg->delta_lf_v = dav1d_get_sbits(gb, 7);
  766|  3.55k|                    hdr->segmentation.seg_data.last_active_segid = i;
  767|  3.55k|                }
  768|  39.0k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (768:21): [True: 3.60k, False: 35.3k]
  ------------------
  769|  3.60k|                    seg->ref = dav1d_get_bits(gb, 3);
  770|  3.60k|                    hdr->segmentation.seg_data.last_active_segid = i;
  771|  3.60k|                    hdr->segmentation.seg_data.preskip = 1;
  772|  35.3k|                } else {
  773|  35.3k|                    seg->ref = -1;
  774|  35.3k|                }
  775|  39.0k|                if ((seg->skip = dav1d_get_bit(gb))) {
  ------------------
  |  Branch (775:21): [True: 4.62k, False: 34.3k]
  ------------------
  776|  4.62k|                    hdr->segmentation.seg_data.last_active_segid = i;
  777|  4.62k|                    hdr->segmentation.seg_data.preskip = 1;
  778|  4.62k|                }
  779|  39.0k|                if ((seg->globalmv = dav1d_get_bit(gb))) {
  ------------------
  |  Branch (779:21): [True: 4.24k, False: 34.7k]
  ------------------
  780|  4.24k|                    hdr->segmentation.seg_data.last_active_segid = i;
  781|  4.24k|                    hdr->segmentation.seg_data.preskip = 1;
  782|  4.24k|                }
  783|  39.0k|            }
  784|  8.16k|        } else {
  785|       |            // segmentation.update_data was false so we should copy
  786|       |            // segmentation data from the reference frame.
  787|  8.16k|            assert(hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (787:13): [True: 8.16k, False: 0]
  ------------------
  788|  8.16k|            const int pri_ref = hdr->refidx[hdr->primary_ref_frame];
  789|  8.16k|            if (!c->refs[pri_ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (789:17): [True: 216, False: 7.95k]
  ------------------
  790|  7.95k|            hdr->segmentation.seg_data =
  791|  7.95k|                c->refs[pri_ref].p.p.frame_hdr->segmentation.seg_data;
  792|  7.95k|        }
  793|  55.6k|    } else {
  794|   500k|        for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++)
  ------------------
  |  |   43|   500k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (794:25): [True: 445k, False: 55.6k]
  ------------------
  795|   445k|            hdr->segmentation.seg_data.d[i].ref = -1;
  796|  55.6k|    }
  797|       |#if DEBUG_FRAME_HDR
  798|       |    printf("HDR: post-segmentation: off=%td\n",
  799|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  800|       |#endif
  801|       |
  802|       |    // delta q
  803|  68.4k|    if (hdr->quant.yac) {
  ------------------
  |  Branch (803:9): [True: 49.3k, False: 19.1k]
  ------------------
  804|  49.3k|        hdr->delta.q.present = dav1d_get_bit(gb);
  805|  49.3k|        if (hdr->delta.q.present) {
  ------------------
  |  Branch (805:13): [True: 7.29k, False: 42.0k]
  ------------------
  806|  7.29k|            hdr->delta.q.res_log2 = dav1d_get_bits(gb, 2);
  807|  7.29k|            if (!hdr->allow_intrabc) {
  ------------------
  |  Branch (807:17): [True: 3.93k, False: 3.35k]
  ------------------
  808|  3.93k|                hdr->delta.lf.present = dav1d_get_bit(gb);
  809|  3.93k|                if (hdr->delta.lf.present) {
  ------------------
  |  Branch (809:21): [True: 1.73k, False: 2.20k]
  ------------------
  810|  1.73k|                    hdr->delta.lf.res_log2 = dav1d_get_bits(gb, 2);
  811|  1.73k|                    hdr->delta.lf.multi = dav1d_get_bit(gb);
  812|  1.73k|                }
  813|  3.93k|            }
  814|  7.29k|        }
  815|  49.3k|    }
  816|       |#if DEBUG_FRAME_HDR
  817|       |    printf("HDR: post-delta_q_lf_flags: off=%td\n",
  818|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  819|       |#endif
  820|       |
  821|       |    // derive lossless flags
  822|  68.4k|    const int delta_lossless = !hdr->quant.ydc_delta && !hdr->quant.udc_delta &&
  ------------------
  |  Branch (822:32): [True: 61.1k, False: 7.32k]
  |  Branch (822:57): [True: 59.5k, False: 1.63k]
  ------------------
  823|  59.5k|        !hdr->quant.uac_delta && !hdr->quant.vdc_delta && !hdr->quant.vac_delta;
  ------------------
  |  Branch (823:9): [True: 58.6k, False: 863]
  |  Branch (823:34): [True: 57.6k, False: 979]
  |  Branch (823:59): [True: 57.6k, False: 38]
  ------------------
  824|  68.4k|    hdr->all_lossless = 1;
  825|   616k|    for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++) {
  ------------------
  |  |   43|   616k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (825:21): [True: 547k, False: 68.4k]
  ------------------
  826|   547k|        hdr->segmentation.qidx[i] = hdr->segmentation.enabled ?
  ------------------
  |  Branch (826:37): [True: 102k, False: 445k]
  ------------------
  827|   102k|            iclip_u8(hdr->quant.yac + hdr->segmentation.seg_data.d[i].delta_q) :
  828|   547k|            hdr->quant.yac;
  829|   547k|        hdr->segmentation.lossless[i] =
  830|   547k|            !hdr->segmentation.qidx[i] && delta_lossless;
  ------------------
  |  Branch (830:13): [True: 153k, False: 394k]
  |  Branch (830:43): [True: 149k, False: 3.71k]
  ------------------
  831|   547k|        hdr->all_lossless &= hdr->segmentation.lossless[i];
  832|   547k|    }
  833|       |
  834|       |    // loopfilter
  835|  68.4k|    if (hdr->all_lossless || hdr->allow_intrabc) {
  ------------------
  |  Branch (835:9): [True: 18.2k, False: 50.1k]
  |  Branch (835:30): [True: 21.3k, False: 28.8k]
  ------------------
  836|  39.6k|        hdr->loopfilter.mode_ref_delta_enabled = 1;
  837|  39.6k|        hdr->loopfilter.mode_ref_delta_update = 1;
  838|  39.6k|        hdr->loopfilter.mode_ref_deltas = default_mode_ref_deltas;
  839|  39.6k|    } else {
  840|  28.8k|        hdr->loopfilter.level_y[0] = dav1d_get_bits(gb, 6);
  841|  28.8k|        hdr->loopfilter.level_y[1] = dav1d_get_bits(gb, 6);
  842|  28.8k|        if (!seqhdr->monochrome &&
  ------------------
  |  Branch (842:13): [True: 10.3k, False: 18.5k]
  ------------------
  843|  10.3k|            (hdr->loopfilter.level_y[0] || hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (843:14): [True: 3.99k, False: 6.33k]
  |  Branch (843:44): [True: 1.17k, False: 5.15k]
  ------------------
  844|  5.17k|        {
  845|  5.17k|            hdr->loopfilter.level_u = dav1d_get_bits(gb, 6);
  846|  5.17k|            hdr->loopfilter.level_v = dav1d_get_bits(gb, 6);
  847|  5.17k|        }
  848|  28.8k|        hdr->loopfilter.sharpness = dav1d_get_bits(gb, 3);
  849|       |
  850|  28.8k|        if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  28.8k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (850:13): [True: 15.1k, False: 13.6k]
  ------------------
  851|  15.1k|            hdr->loopfilter.mode_ref_deltas = default_mode_ref_deltas;
  852|  15.1k|        } else {
  853|  13.6k|            const int ref = hdr->refidx[hdr->primary_ref_frame];
  854|  13.6k|            if (!c->refs[ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (854:17): [True: 479, False: 13.2k]
  ------------------
  855|  13.2k|            hdr->loopfilter.mode_ref_deltas =
  856|  13.2k|                c->refs[ref].p.p.frame_hdr->loopfilter.mode_ref_deltas;
  857|  13.2k|        }
  858|  28.3k|        hdr->loopfilter.mode_ref_delta_enabled = dav1d_get_bit(gb);
  859|  28.3k|        if (hdr->loopfilter.mode_ref_delta_enabled) {
  ------------------
  |  Branch (859:13): [True: 15.3k, False: 13.0k]
  ------------------
  860|  15.3k|            hdr->loopfilter.mode_ref_delta_update = dav1d_get_bit(gb);
  861|  15.3k|            if (hdr->loopfilter.mode_ref_delta_update) {
  ------------------
  |  Branch (861:17): [True: 1.69k, False: 13.6k]
  ------------------
  862|  15.2k|                for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (862:33): [True: 13.5k, False: 1.69k]
  ------------------
  863|  13.5k|                    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (863:25): [True: 3.88k, False: 9.65k]
  ------------------
  864|  3.88k|                        hdr->loopfilter.mode_ref_deltas.ref_delta[i] =
  865|  3.88k|                            dav1d_get_sbits(gb, 7);
  866|  5.07k|                for (int i = 0; i < 2; i++)
  ------------------
  |  Branch (866:33): [True: 3.38k, False: 1.69k]
  ------------------
  867|  3.38k|                    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (867:25): [True: 878, False: 2.50k]
  ------------------
  868|    878|                        hdr->loopfilter.mode_ref_deltas.mode_delta[i] =
  869|    878|                            dav1d_get_sbits(gb, 7);
  870|  1.69k|            }
  871|  15.3k|        }
  872|  28.3k|    }
  873|       |#if DEBUG_FRAME_HDR
  874|       |    printf("HDR: post-lpf: off=%td\n",
  875|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  876|       |#endif
  877|       |
  878|       |    // cdef
  879|  68.0k|    if (!hdr->all_lossless && seqhdr->cdef && !hdr->allow_intrabc) {
  ------------------
  |  Branch (879:9): [True: 49.7k, False: 18.2k]
  |  Branch (879:31): [True: 18.7k, False: 30.9k]
  |  Branch (879:47): [True: 7.86k, False: 10.9k]
  ------------------
  880|  7.86k|        hdr->cdef.damping = dav1d_get_bits(gb, 2) + 3;
  881|  7.86k|        hdr->cdef.n_bits = dav1d_get_bits(gb, 2);
  882|  19.3k|        for (int i = 0; i < (1 << hdr->cdef.n_bits); i++) {
  ------------------
  |  Branch (882:25): [True: 11.5k, False: 7.86k]
  ------------------
  883|  11.5k|            hdr->cdef.y_strength[i] = dav1d_get_bits(gb, 6);
  884|  11.5k|            if (!seqhdr->monochrome)
  ------------------
  |  Branch (884:17): [True: 7.78k, False: 3.72k]
  ------------------
  885|  7.78k|                hdr->cdef.uv_strength[i] = dav1d_get_bits(gb, 6);
  886|  11.5k|        }
  887|  7.86k|    }
  888|       |#if DEBUG_FRAME_HDR
  889|       |    printf("HDR: post-cdef: off=%td\n",
  890|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  891|       |#endif
  892|       |
  893|       |    // restoration
  894|  68.0k|    if ((!hdr->all_lossless || hdr->super_res.enabled) &&
  ------------------
  |  Branch (894:10): [True: 49.7k, False: 18.2k]
  |  Branch (894:32): [True: 5.12k, False: 13.1k]
  ------------------
  895|  54.8k|        seqhdr->restoration && !hdr->allow_intrabc)
  ------------------
  |  Branch (895:9): [True: 35.9k, False: 18.8k]
  |  Branch (895:32): [True: 26.3k, False: 9.64k]
  ------------------
  896|  26.3k|    {
  897|  26.3k|        hdr->restoration.type[0] = dav1d_get_bits(gb, 2);
  898|  26.3k|        if (!seqhdr->monochrome) {
  ------------------
  |  Branch (898:13): [True: 9.92k, False: 16.3k]
  ------------------
  899|  9.92k|            hdr->restoration.type[1] = dav1d_get_bits(gb, 2);
  900|  9.92k|            hdr->restoration.type[2] = dav1d_get_bits(gb, 2);
  901|  9.92k|        }
  902|       |
  903|  26.3k|        if (hdr->restoration.type[0] || hdr->restoration.type[1] ||
  ------------------
  |  Branch (903:13): [True: 18.6k, False: 7.71k]
  |  Branch (903:41): [True: 465, False: 7.25k]
  ------------------
  904|  7.25k|            hdr->restoration.type[2])
  ------------------
  |  Branch (904:13): [True: 578, False: 6.67k]
  ------------------
  905|  19.6k|        {
  906|       |            // Log2 of the restoration unit size.
  907|  19.6k|            hdr->restoration.unit_size[0] = 6 + seqhdr->sb128;
  908|  19.6k|            if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (908:17): [True: 13.6k, False: 5.97k]
  ------------------
  909|  13.6k|                hdr->restoration.unit_size[0]++;
  910|  13.6k|                if (!seqhdr->sb128)
  ------------------
  |  Branch (910:21): [True: 2.59k, False: 11.0k]
  ------------------
  911|  2.59k|                    hdr->restoration.unit_size[0] += dav1d_get_bit(gb);
  912|  13.6k|            }
  913|  19.6k|            hdr->restoration.unit_size[1] = hdr->restoration.unit_size[0];
  914|  19.6k|            if ((hdr->restoration.type[1] || hdr->restoration.type[2]) &&
  ------------------
  |  Branch (914:18): [True: 3.10k, False: 16.5k]
  |  Branch (914:46): [True: 3.06k, False: 13.4k]
  ------------------
  915|  6.16k|                seqhdr->ss_hor == 1 && seqhdr->ss_ver == 1)
  ------------------
  |  Branch (915:17): [True: 4.12k, False: 2.04k]
  |  Branch (915:40): [True: 2.28k, False: 1.83k]
  ------------------
  916|  2.28k|            {
  917|  2.28k|                hdr->restoration.unit_size[1] -= dav1d_get_bit(gb);
  918|  2.28k|            }
  919|  19.6k|        } else {
  920|  6.67k|            hdr->restoration.unit_size[0] = 8;
  921|  6.67k|        }
  922|  26.3k|    }
  923|       |#if DEBUG_FRAME_HDR
  924|       |    printf("HDR: post-restoration: off=%td\n",
  925|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  926|       |#endif
  927|       |
  928|  68.0k|    if (!hdr->all_lossless)
  ------------------
  |  Branch (928:9): [True: 49.7k, False: 18.2k]
  ------------------
  929|  49.7k|        hdr->txfm_mode = dav1d_get_bit(gb) ? DAV1D_TX_SWITCHABLE : DAV1D_TX_LARGEST;
  ------------------
  |  Branch (929:26): [True: 9.04k, False: 40.6k]
  ------------------
  930|       |#if DEBUG_FRAME_HDR
  931|       |    printf("HDR: post-txfmmode: off=%td\n",
  932|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  933|       |#endif
  934|  68.0k|    if (IS_INTER_OR_SWITCH(hdr))
  ------------------
  |  |   36|  68.0k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 22.6k, False: 45.3k]
  |  |  ------------------
  ------------------
  935|  22.6k|        hdr->switchable_comp_refs = dav1d_get_bit(gb);
  936|       |#if DEBUG_FRAME_HDR
  937|       |    printf("HDR: post-refmode: off=%td\n",
  938|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  939|       |#endif
  940|  68.0k|    if (hdr->switchable_comp_refs && IS_INTER_OR_SWITCH(hdr) && seqhdr->order_hint) {
  ------------------
  |  |   36|  81.4k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 13.4k, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (940:9): [True: 13.4k, False: 54.5k]
  |  Branch (940:65): [True: 11.7k, False: 1.78k]
  ------------------
  941|  11.7k|        const int poc = hdr->frame_offset;
  942|  11.7k|        int off_before = -1, off_after = -1;
  943|  11.7k|        int off_before_idx, off_after_idx;
  944|  92.4k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (944:25): [True: 80.9k, False: 11.5k]
  ------------------
  945|  80.9k|            if (!c->refs[hdr->refidx[i]].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (945:17): [True: 198, False: 80.7k]
  ------------------
  946|  80.7k|            const int refpoc = c->refs[hdr->refidx[i]].p.p.frame_hdr->frame_offset;
  947|       |
  948|  80.7k|            const int diff = get_poc_diff(seqhdr->order_hint_n_bits, refpoc, poc);
  949|  80.7k|            if (diff > 0) {
  ------------------
  |  Branch (949:17): [True: 14.8k, False: 65.9k]
  ------------------
  950|  14.8k|                if (off_after < 0 || get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (950:21): [True: 5.64k, False: 9.15k]
  |  Branch (950:38): [True: 338, False: 8.82k]
  ------------------
  951|  9.15k|                                                  off_after, refpoc) > 0)
  952|  5.98k|                {
  953|  5.98k|                    off_after = refpoc;
  954|  5.98k|                    off_after_idx = i;
  955|  5.98k|                }
  956|  65.9k|            } else if (diff < 0 && (off_before < 0 ||
  ------------------
  |  Branch (956:24): [True: 23.9k, False: 42.0k]
  |  Branch (956:37): [True: 5.56k, False: 18.4k]
  ------------------
  957|  18.4k|                                    get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (957:37): [True: 922, False: 17.4k]
  ------------------
  958|  18.4k|                                                 refpoc, off_before) > 0))
  959|  6.48k|            {
  960|  6.48k|                off_before = refpoc;
  961|  6.48k|                off_before_idx = i;
  962|  6.48k|            }
  963|  80.7k|        }
  964|       |
  965|  11.5k|        if ((off_before | off_after) >= 0) {
  ------------------
  |  Branch (965:13): [True: 1.39k, False: 10.1k]
  ------------------
  966|  1.39k|            hdr->skip_mode_refs[0] = imin(off_before_idx, off_after_idx);
  967|  1.39k|            hdr->skip_mode_refs[1] = imax(off_before_idx, off_after_idx);
  968|  1.39k|            hdr->skip_mode_allowed = 1;
  969|  10.1k|        } else if (off_before >= 0) {
  ------------------
  |  Branch (969:20): [True: 4.03k, False: 6.07k]
  ------------------
  970|  4.03k|            int off_before2 = -1;
  971|  4.03k|            int off_before2_idx;
  972|  32.2k|            for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (972:29): [True: 28.2k, False: 4.03k]
  ------------------
  973|  28.2k|                if (!c->refs[hdr->refidx[i]].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (973:21): [True: 0, False: 28.2k]
  ------------------
  974|  28.2k|                const int refpoc = c->refs[hdr->refidx[i]].p.p.frame_hdr->frame_offset;
  975|  28.2k|                if (get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (975:21): [True: 7.84k, False: 20.3k]
  ------------------
  976|  28.2k|                                 refpoc, off_before) < 0) {
  977|  7.84k|                    if (off_before2 < 0 || get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (977:25): [True: 1.70k, False: 6.13k]
  |  Branch (977:44): [True: 318, False: 5.81k]
  ------------------
  978|  6.13k|                                                        refpoc, off_before2) > 0)
  979|  2.02k|                    {
  980|  2.02k|                        off_before2 = refpoc;
  981|  2.02k|                        off_before2_idx = i;
  982|  2.02k|                    }
  983|  7.84k|                }
  984|  28.2k|            }
  985|       |
  986|  4.03k|            if (off_before2 >= 0) {
  ------------------
  |  Branch (986:17): [True: 1.70k, False: 2.32k]
  ------------------
  987|  1.70k|                hdr->skip_mode_refs[0] = imin(off_before_idx, off_before2_idx);
  988|  1.70k|                hdr->skip_mode_refs[1] = imax(off_before_idx, off_before2_idx);
  989|  1.70k|                hdr->skip_mode_allowed = 1;
  990|  1.70k|            }
  991|  4.03k|        }
  992|  11.5k|    }
  993|  67.8k|    if (hdr->skip_mode_allowed)
  ------------------
  |  Branch (993:9): [True: 3.10k, False: 64.7k]
  ------------------
  994|  3.10k|        hdr->skip_mode_enabled = dav1d_get_bit(gb);
  995|       |#if DEBUG_FRAME_HDR
  996|       |    printf("HDR: post-extskip: off=%td\n",
  997|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  998|       |#endif
  999|  67.8k|    if (!hdr->error_resilient_mode && IS_INTER_OR_SWITCH(hdr) && seqhdr->warped_motion)
  ------------------
  |  |   36|  89.2k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 19.7k, False: 1.69k]
  |  |  ------------------
  ------------------
  |  Branch (999:9): [True: 21.4k, False: 46.3k]
  |  Branch (999:66): [True: 13.6k, False: 6.10k]
  ------------------
 1000|  13.6k|        hdr->warp_motion = dav1d_get_bit(gb);
 1001|       |#if DEBUG_FRAME_HDR
 1002|       |    printf("HDR: post-warpmotionbit: off=%td\n",
 1003|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1004|       |#endif
 1005|  67.8k|    hdr->reduced_txtp_set = dav1d_get_bit(gb);
 1006|       |#if DEBUG_FRAME_HDR
 1007|       |    printf("HDR: post-reducedtxtpset: off=%td\n",
 1008|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1009|       |#endif
 1010|       |
 1011|   542k|    for (int i = 0; i < 7; i++)
  ------------------
  |  Branch (1011:21): [True: 474k, False: 67.8k]
  ------------------
 1012|   474k|        hdr->gmv[i] = dav1d_default_wm_params;
 1013|       |
 1014|  67.8k|    if (IS_INTER_OR_SWITCH(hdr)) {
  ------------------
  |  |   36|  67.8k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 22.4k, False: 45.3k]
  |  |  ------------------
  ------------------
 1015|   178k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (1015:25): [True: 156k, False: 22.2k]
  ------------------
 1016|   156k|            hdr->gmv[i].type = !dav1d_get_bit(gb) ? DAV1D_WM_TYPE_IDENTITY :
  ------------------
  |  Branch (1016:32): [True: 146k, False: 9.94k]
  ------------------
 1017|   156k|                                dav1d_get_bit(gb) ? DAV1D_WM_TYPE_ROT_ZOOM :
  ------------------
  |  Branch (1017:33): [True: 5.91k, False: 4.03k]
  ------------------
 1018|  9.94k|                                dav1d_get_bit(gb) ? DAV1D_WM_TYPE_TRANSLATION :
  ------------------
  |  Branch (1018:33): [True: 1.91k, False: 2.11k]
  ------------------
 1019|  4.03k|                                                    DAV1D_WM_TYPE_AFFINE;
 1020|       |
 1021|   156k|            if (hdr->gmv[i].type == DAV1D_WM_TYPE_IDENTITY) continue;
  ------------------
  |  Branch (1021:17): [True: 146k, False: 9.94k]
  ------------------
 1022|       |
 1023|  9.94k|            const Dav1dWarpedMotionParams *ref_gmv;
 1024|  9.94k|            if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  9.94k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (1024:17): [True: 1.63k, False: 8.31k]
  ------------------
 1025|  1.63k|                ref_gmv = &dav1d_default_wm_params;
 1026|  8.31k|            } else {
 1027|  8.31k|                const int pri_ref = hdr->refidx[hdr->primary_ref_frame];
 1028|  8.31k|                if (!c->refs[pri_ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1028:21): [True: 203, False: 8.11k]
  ------------------
 1029|  8.11k|                ref_gmv = &c->refs[pri_ref].p.p.frame_hdr->gmv[i];
 1030|  8.11k|            }
 1031|  9.74k|            int32_t *const mat = hdr->gmv[i].matrix;
 1032|  9.74k|            const int32_t *const ref_mat = ref_gmv->matrix;
 1033|  9.74k|            int bits, shift;
 1034|       |
 1035|  9.74k|            if (hdr->gmv[i].type >= DAV1D_WM_TYPE_ROT_ZOOM) {
  ------------------
  |  Branch (1035:17): [True: 7.82k, False: 1.91k]
  ------------------
 1036|  7.82k|                mat[2] = (1 << 16) + 2 *
 1037|  7.82k|                    dav1d_get_bits_subexp(gb, (ref_mat[2] - (1 << 16)) >> 1, 12);
 1038|  7.82k|                mat[3] = 2 * dav1d_get_bits_subexp(gb, ref_mat[3] >> 1, 12);
 1039|       |
 1040|  7.82k|                bits = 12;
 1041|  7.82k|                shift = 10;
 1042|  7.82k|            } else {
 1043|  1.91k|                bits = 9 - !hdr->hp;
 1044|  1.91k|                shift = 13 + !hdr->hp;
 1045|  1.91k|            }
 1046|       |
 1047|  9.74k|            if (hdr->gmv[i].type == DAV1D_WM_TYPE_AFFINE) {
  ------------------
  |  Branch (1047:17): [True: 1.97k, False: 7.77k]
  ------------------
 1048|  1.97k|                mat[4] = 2 * dav1d_get_bits_subexp(gb, ref_mat[4] >> 1, 12);
 1049|  1.97k|                mat[5] = (1 << 16) + 2 *
 1050|  1.97k|                    dav1d_get_bits_subexp(gb, (ref_mat[5] - (1 << 16)) >> 1, 12);
 1051|  7.77k|            } else {
 1052|  7.77k|                mat[4] = -mat[3];
 1053|  7.77k|                mat[5] = mat[2];
 1054|  7.77k|            }
 1055|       |
 1056|  9.74k|            mat[0] = dav1d_get_bits_subexp(gb, ref_mat[0] >> shift, bits) * (1 << shift);
 1057|  9.74k|            mat[1] = dav1d_get_bits_subexp(gb, ref_mat[1] >> shift, bits) * (1 << shift);
 1058|  9.74k|        }
 1059|  22.4k|    }
 1060|       |#if DEBUG_FRAME_HDR
 1061|       |    printf("HDR: post-gmv: off=%td\n",
 1062|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1063|       |#endif
 1064|       |
 1065|  67.6k|    if (seqhdr->film_grain_present && (hdr->show_frame || hdr->showable_frame)) {
  ------------------
  |  Branch (1065:9): [True: 20.6k, False: 46.9k]
  |  Branch (1065:40): [True: 15.6k, False: 4.95k]
  |  Branch (1065:59): [True: 3.69k, False: 1.25k]
  ------------------
 1066|  19.3k|        hdr->film_grain.present = dav1d_get_bit(gb);
 1067|  19.3k|        if (hdr->film_grain.present) {
  ------------------
  |  Branch (1067:13): [True: 6.09k, False: 13.2k]
  ------------------
 1068|  6.09k|            const unsigned seed = dav1d_get_bits(gb, 16);
 1069|  6.09k|            hdr->film_grain.update = hdr->frame_type != DAV1D_FRAME_TYPE_INTER || dav1d_get_bit(gb);
  ------------------
  |  Branch (1069:38): [True: 4.41k, False: 1.68k]
  |  Branch (1069:83): [True: 67, False: 1.62k]
  ------------------
 1070|  6.09k|            if (!hdr->film_grain.update) {
  ------------------
  |  Branch (1070:17): [True: 1.62k, False: 4.47k]
  ------------------
 1071|  1.62k|                const int refidx = dav1d_get_bits(gb, 3);
 1072|  1.62k|                int i;
 1073|  7.34k|                for (i = 0; i < 7; i++)
  ------------------
  |  Branch (1073:29): [True: 7.12k, False: 224]
  ------------------
 1074|  7.12k|                    if (hdr->refidx[i] == refidx)
  ------------------
  |  Branch (1074:25): [True: 1.39k, False: 5.72k]
  ------------------
 1075|  1.39k|                        break;
 1076|  1.62k|                if (i == 7 || !c->refs[refidx].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1076:21): [True: 224, False: 1.39k]
  |  Branch (1076:31): [True: 284, False: 1.11k]
  ------------------
 1077|  1.11k|                hdr->film_grain.data = c->refs[refidx].p.p.frame_hdr->film_grain.data;
 1078|  1.11k|                hdr->film_grain.data.seed = seed;
 1079|  4.47k|            } else {
 1080|  4.47k|                Dav1dFilmGrainData *const fgd = &hdr->film_grain.data;
 1081|  4.47k|                fgd->seed = seed;
 1082|       |
 1083|  4.47k|                fgd->num_y_points = dav1d_get_bits(gb, 4);
 1084|  4.47k|                if (fgd->num_y_points > 14) goto error;
  ------------------
  |  Branch (1084:21): [True: 210, False: 4.26k]
  ------------------
 1085|  7.04k|                for (int i = 0; i < fgd->num_y_points; i++) {
  ------------------
  |  Branch (1085:33): [True: 3.33k, False: 3.70k]
  ------------------
 1086|  3.33k|                    fgd->y_points[i][0] = dav1d_get_bits(gb, 8);
 1087|  3.33k|                    if (i && fgd->y_points[i - 1][0] >= fgd->y_points[i][0])
  ------------------
  |  Branch (1087:25): [True: 1.10k, False: 2.23k]
  |  Branch (1087:30): [True: 562, False: 542]
  ------------------
 1088|    562|                        goto error;
 1089|  2.77k|                    fgd->y_points[i][1] = dav1d_get_bits(gb, 8);
 1090|  2.77k|                }
 1091|       |
 1092|  3.70k|                if (!seqhdr->monochrome)
  ------------------
  |  Branch (1092:21): [True: 3.23k, False: 475]
  ------------------
 1093|  3.23k|                    fgd->chroma_scaling_from_luma = dav1d_get_bit(gb);
 1094|  3.70k|                if (seqhdr->monochrome || fgd->chroma_scaling_from_luma ||
  ------------------
  |  Branch (1094:21): [True: 475, False: 3.23k]
  |  Branch (1094:43): [True: 643, False: 2.58k]
  ------------------
 1095|  2.58k|                    (seqhdr->ss_ver == 1 && seqhdr->ss_hor == 1 && !fgd->num_y_points))
  ------------------
  |  Branch (1095:22): [True: 1.21k, False: 1.36k]
  |  Branch (1095:45): [True: 1.21k, False: 0]
  |  Branch (1095:68): [True: 209, False: 1.01k]
  ------------------
 1096|  1.32k|                {
 1097|  1.32k|                    fgd->num_uv_points[0] = fgd->num_uv_points[1] = 0;
 1098|  5.72k|                } else for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1098:41): [True: 4.13k, False: 1.58k]
  ------------------
 1099|  4.13k|                    fgd->num_uv_points[pl] = dav1d_get_bits(gb, 4);
 1100|  4.13k|                    if (fgd->num_uv_points[pl] > 10) goto error;
  ------------------
  |  Branch (1100:25): [True: 214, False: 3.92k]
  ------------------
 1101|  6.45k|                    for (int i = 0; i < fgd->num_uv_points[pl]; i++) {
  ------------------
  |  Branch (1101:37): [True: 3.11k, False: 3.34k]
  ------------------
 1102|  3.11k|                        fgd->uv_points[pl][i][0] = dav1d_get_bits(gb, 8);
 1103|  3.11k|                        if (i && fgd->uv_points[pl][i - 1][0] >= fgd->uv_points[pl][i][0])
  ------------------
  |  Branch (1103:29): [True: 1.61k, False: 1.50k]
  |  Branch (1103:34): [True: 581, False: 1.02k]
  ------------------
 1104|    581|                            goto error;
 1105|  2.52k|                        fgd->uv_points[pl][i][1] = dav1d_get_bits(gb, 8);
 1106|  2.52k|                    }
 1107|  3.92k|                }
 1108|       |
 1109|  2.91k|                if (seqhdr->ss_hor == 1 && seqhdr->ss_ver == 1 &&
  ------------------
  |  Branch (1109:21): [True: 2.40k, False: 507]
  |  Branch (1109:44): [True: 1.35k, False: 1.05k]
  ------------------
 1110|  1.35k|                    !!fgd->num_uv_points[0] != !!fgd->num_uv_points[1])
  ------------------
  |  Branch (1110:21): [True: 66, False: 1.28k]
  ------------------
 1111|     66|                {
 1112|     66|                    goto error;
 1113|     66|                }
 1114|       |
 1115|  2.84k|                fgd->scaling_shift = dav1d_get_bits(gb, 2) + 8;
 1116|  2.84k|                fgd->ar_coeff_lag = dav1d_get_bits(gb, 2);
 1117|  2.84k|                const int num_y_pos = 2 * fgd->ar_coeff_lag * (fgd->ar_coeff_lag + 1);
 1118|  2.84k|                if (fgd->num_y_points)
  ------------------
  |  Branch (1118:21): [True: 1.22k, False: 1.62k]
  ------------------
 1119|  7.04k|                    for (int i = 0; i < num_y_pos; i++)
  ------------------
  |  Branch (1119:37): [True: 5.82k, False: 1.22k]
  ------------------
 1120|  5.82k|                        fgd->ar_coeffs_y[i] = dav1d_get_bits(gb, 8) - 128;
 1121|  8.53k|                for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (1121:34): [True: 5.69k, False: 2.84k]
  ------------------
 1122|  5.69k|                    if (fgd->num_uv_points[pl] || fgd->chroma_scaling_from_luma) {
  ------------------
  |  Branch (1122:25): [True: 752, False: 4.93k]
  |  Branch (1122:51): [True: 1.28k, False: 3.65k]
  ------------------
 1123|  2.03k|                        const int num_uv_pos = num_y_pos + !!fgd->num_y_points;
 1124|  21.3k|                        for (int i = 0; i < num_uv_pos; i++)
  ------------------
  |  Branch (1124:41): [True: 19.3k, False: 2.03k]
  ------------------
 1125|  19.3k|                            fgd->ar_coeffs_uv[pl][i] = dav1d_get_bits(gb, 8) - 128;
 1126|  2.03k|                        if (!fgd->num_y_points)
  ------------------
  |  Branch (1126:29): [True: 1.28k, False: 758]
  ------------------
 1127|  1.28k|                            fgd->ar_coeffs_uv[pl][num_uv_pos] = 0;
 1128|  2.03k|                    }
 1129|  2.84k|                fgd->ar_coeff_shift = dav1d_get_bits(gb, 2) + 6;
 1130|  2.84k|                fgd->grain_scale_shift = dav1d_get_bits(gb, 2);
 1131|  8.53k|                for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (1131:34): [True: 5.69k, False: 2.84k]
  ------------------
 1132|  5.69k|                    if (fgd->num_uv_points[pl]) {
  ------------------
  |  Branch (1132:25): [True: 752, False: 4.93k]
  ------------------
 1133|    752|                        fgd->uv_mult[pl] = dav1d_get_bits(gb, 8) - 128;
 1134|    752|                        fgd->uv_luma_mult[pl] = dav1d_get_bits(gb, 8) - 128;
 1135|    752|                        fgd->uv_offset[pl] = dav1d_get_bits(gb, 9) - 256;
 1136|    752|                    }
 1137|  2.84k|                fgd->overlap_flag = dav1d_get_bit(gb);
 1138|  2.84k|                fgd->clip_to_restricted_range = dav1d_get_bit(gb);
 1139|  2.84k|            }
 1140|  6.09k|        }
 1141|  19.3k|    }
 1142|       |#if DEBUG_FRAME_HDR
 1143|       |    printf("HDR: post-filmgrain: off=%td\n",
 1144|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1145|       |#endif
 1146|       |
 1147|  65.4k|    return 0;
 1148|       |
 1149|  5.35k|error:
 1150|  5.35k|    dav1d_log(c, "Error parsing frame header\n");
  ------------------
  |  |   44|  5.35k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 5.35k]
  |  |  ------------------
  ------------------
 1151|  5.35k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  5.35k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1152|  67.6k|}
obu.c:read_frame_size:
  343|  69.1k|{
  344|  69.1k|    const Dav1dSequenceHeader *const seqhdr = c->seq_hdr;
  345|  69.1k|    Dav1dFrameHeader *const hdr = c->frame_hdr;
  346|       |
  347|  69.1k|    if (use_ref) {
  ------------------
  |  Branch (347:9): [True: 12.3k, False: 56.7k]
  ------------------
  348|  25.2k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (348:25): [True: 23.9k, False: 1.27k]
  ------------------
  349|  23.9k|            if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (349:17): [True: 11.0k, False: 12.8k]
  ------------------
  350|  11.0k|                const Dav1dThreadPicture *const ref =
  351|  11.0k|                    &c->refs[c->frame_hdr->refidx[i]].p;
  352|  11.0k|                if (!ref->p.frame_hdr) return -1;
  ------------------
  |  Branch (352:21): [True: 242, False: 10.8k]
  ------------------
  353|  10.8k|                hdr->width[1] = ref->p.frame_hdr->width[1];
  354|  10.8k|                hdr->height = ref->p.frame_hdr->height;
  355|  10.8k|                hdr->render_width = ref->p.frame_hdr->render_width;
  356|  10.8k|                hdr->render_height = ref->p.frame_hdr->render_height;
  357|  10.8k|                hdr->super_res.enabled = seqhdr->super_res && dav1d_get_bit(gb);
  ------------------
  |  Branch (357:42): [True: 1.31k, False: 9.54k]
  |  Branch (357:63): [True: 965, False: 348]
  ------------------
  358|  10.8k|                if (hdr->super_res.enabled) {
  ------------------
  |  Branch (358:21): [True: 965, False: 9.89k]
  ------------------
  359|    965|                    const int d = hdr->super_res.width_scale_denominator =
  360|    965|                        9 + dav1d_get_bits(gb, 3);
  361|    965|                    hdr->width[0] = imax((hdr->width[1] * 8 + (d >> 1)) / d,
  362|    965|                                         imin(16, hdr->width[1]));
  363|  9.89k|                } else {
  364|  9.89k|                    hdr->super_res.width_scale_denominator = 8;
  365|  9.89k|                    hdr->width[0] = hdr->width[1];
  366|  9.89k|                }
  367|  10.8k|                return 0;
  368|  11.0k|            }
  369|  23.9k|        }
  370|  12.3k|    }
  371|       |
  372|  58.0k|    if (hdr->frame_size_override) {
  ------------------
  |  Branch (372:9): [True: 9.42k, False: 48.6k]
  ------------------
  373|  9.42k|        hdr->width[1] = dav1d_get_bits(gb, seqhdr->width_n_bits) + 1;
  374|  9.42k|        hdr->height = dav1d_get_bits(gb, seqhdr->height_n_bits) + 1;
  375|  48.6k|    } else {
  376|  48.6k|        hdr->width[1] = seqhdr->max_width;
  377|  48.6k|        hdr->height = seqhdr->max_height;
  378|  48.6k|    }
  379|  58.0k|    hdr->super_res.enabled = seqhdr->super_res && dav1d_get_bit(gb);
  ------------------
  |  Branch (379:30): [True: 33.3k, False: 24.7k]
  |  Branch (379:51): [True: 8.21k, False: 25.0k]
  ------------------
  380|  58.0k|    if (hdr->super_res.enabled) {
  ------------------
  |  Branch (380:9): [True: 8.21k, False: 49.8k]
  ------------------
  381|  8.21k|        const int d = hdr->super_res.width_scale_denominator = 9 + dav1d_get_bits(gb, 3);
  382|  8.21k|        hdr->width[0] = imax((hdr->width[1] * 8 + (d >> 1)) / d, imin(16, hdr->width[1]));
  383|  49.8k|    } else {
  384|  49.8k|        hdr->super_res.width_scale_denominator = 8;
  385|  49.8k|        hdr->width[0] = hdr->width[1];
  386|  49.8k|    }
  387|  58.0k|    hdr->have_render_size = dav1d_get_bit(gb);
  388|  58.0k|    if (hdr->have_render_size) {
  ------------------
  |  Branch (388:9): [True: 2.63k, False: 55.4k]
  ------------------
  389|  2.63k|        hdr->render_width = dav1d_get_bits(gb, 16) + 1;
  390|  2.63k|        hdr->render_height = dav1d_get_bits(gb, 16) + 1;
  391|  55.4k|    } else {
  392|  55.4k|        hdr->render_width = hdr->width[1];
  393|  55.4k|        hdr->render_height = hdr->height;
  394|  55.4k|    }
  395|  58.0k|    return 0;
  396|  69.1k|}
obu.c:tile_log2:
  398|   328k|static inline int tile_log2(const int sz, const int tgt) {
  399|   328k|    int k;
  400|   476k|    for (k = 0; (sz << k) < tgt; k++) ;
  ------------------
  |  Branch (400:17): [True: 148k, False: 328k]
  ------------------
  401|   328k|    return k;
  402|   328k|}
obu.c:check_trailing_bits:
   50|  48.6k|{
   51|  48.6k|    const int trailing_one_bit = dav1d_get_bit(gb);
   52|       |
   53|  48.6k|    if (gb->error)
  ------------------
  |  Branch (53:9): [True: 8.63k, False: 40.0k]
  ------------------
   54|  8.63k|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  8.63k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   55|       |
   56|  40.0k|    if (!strict_std_compliance)
  ------------------
  |  Branch (56:9): [True: 40.0k, False: 0]
  ------------------
   57|  40.0k|        return 0;
   58|       |
   59|      0|    if (!trailing_one_bit || gb->state)
  ------------------
  |  Branch (59:9): [True: 0, False: 0]
  |  Branch (59:30): [True: 0, False: 0]
  ------------------
   60|      0|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   61|       |
   62|      0|    ptrdiff_t size = gb->ptr_end - gb->ptr;
   63|      0|    while (size > 0 && gb->ptr[size - 1] == 0)
  ------------------
  |  Branch (63:12): [True: 0, False: 0]
  |  Branch (63:24): [True: 0, False: 0]
  ------------------
   64|      0|        size--;
   65|       |
   66|      0|    if (size)
  ------------------
  |  Branch (66:9): [True: 0, False: 0]
  ------------------
   67|      0|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   68|       |
   69|      0|    return 0;
   70|      0|}
obu.c:parse_tile_hdr:
 1154|  58.2k|static void parse_tile_hdr(Dav1dContext *const c, GetBits *const gb) {
 1155|  58.2k|    const int n_tiles = c->frame_hdr->tiling.cols * c->frame_hdr->tiling.rows;
 1156|  58.2k|    const int have_tile_pos = n_tiles > 1 ? dav1d_get_bit(gb) : 0;
  ------------------
  |  Branch (1156:31): [True: 6.05k, False: 52.2k]
  ------------------
 1157|       |
 1158|  58.2k|    if (have_tile_pos) {
  ------------------
  |  Branch (1158:9): [True: 2.09k, False: 56.1k]
  ------------------
 1159|  2.09k|        const int n_bits = c->frame_hdr->tiling.log2_cols +
 1160|  2.09k|                           c->frame_hdr->tiling.log2_rows;
 1161|  2.09k|        c->tile[c->n_tile_data].start = dav1d_get_bits(gb, n_bits);
 1162|  2.09k|        c->tile[c->n_tile_data].end = dav1d_get_bits(gb, n_bits);
 1163|  56.1k|    } else {
 1164|  56.1k|        c->tile[c->n_tile_data].start = 0;
 1165|  56.1k|        c->tile[c->n_tile_data].end = n_tiles - 1;
 1166|  56.1k|    }
 1167|  58.2k|}

dav1d_pal_dsp_init:
   71|  10.2k|COLD void dav1d_pal_dsp_init(Dav1dPalDSPContext *const c) {
   72|  10.2k|    c->pal_idx_finish = pal_idx_finish_c;
   73|       |
   74|  10.2k|#if HAVE_ASM
   75|       |#if ARCH_RISCV
   76|       |    pal_dsp_init_riscv(c);
   77|       |#elif ARCH_X86
   78|       |    pal_dsp_init_x86(c);
   79|  10.2k|#endif
   80|  10.2k|#endif
   81|  10.2k|}

dav1d_default_picture_alloc:
   46|  54.9k|int dav1d_default_picture_alloc(Dav1dPicture *const p, void *const cookie) {
   47|  54.9k|    const int hbd = p->p.bpc > 8;
   48|  54.9k|    const int aligned_w = (p->p.w + 127) & ~127;
   49|  54.9k|    const int aligned_h = (p->p.h + 127) & ~127;
   50|  54.9k|    const int has_chroma = p->p.layout != DAV1D_PIXEL_LAYOUT_I400;
   51|  54.9k|    const int ss_ver = p->p.layout == DAV1D_PIXEL_LAYOUT_I420;
   52|  54.9k|    const int ss_hor = p->p.layout != DAV1D_PIXEL_LAYOUT_I444;
   53|  54.9k|    ptrdiff_t y_stride = aligned_w << hbd;
   54|  54.9k|    ptrdiff_t uv_stride = has_chroma ? y_stride >> ss_hor : 0;
  ------------------
  |  Branch (54:27): [True: 29.4k, False: 25.5k]
  ------------------
   55|       |    /* Due to how mapping of addresses to sets works in most L1 and L2 cache
   56|       |     * implementations, strides of multiples of certain power-of-two numbers
   57|       |     * may cause multiple rows of the same superblock to map to the same set,
   58|       |     * causing evictions of previous rows resulting in a reduction in cache
   59|       |     * hit rate. Avoid that by slightly padding the stride when necessary. */
   60|  54.9k|    if (!(y_stride & 1023))
  ------------------
  |  Branch (60:9): [True: 5.62k, False: 49.3k]
  ------------------
   61|  5.62k|        y_stride += DAV1D_PICTURE_ALIGNMENT;
  ------------------
  |  |   44|  5.62k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   62|  54.9k|    if (!(uv_stride & 1023) && has_chroma)
  ------------------
  |  Branch (62:9): [True: 29.4k, False: 25.5k]
  |  Branch (62:32): [True: 3.89k, False: 25.5k]
  ------------------
   63|  3.89k|        uv_stride += DAV1D_PICTURE_ALIGNMENT;
  ------------------
  |  |   44|  3.89k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   64|  54.9k|    p->stride[0] = y_stride;
   65|  54.9k|    p->stride[1] = uv_stride;
   66|  54.9k|    const size_t y_sz = y_stride * aligned_h;
   67|  54.9k|    const size_t uv_sz = uv_stride * (aligned_h >> ss_ver);
   68|  54.9k|    const size_t pic_size = y_sz + 2 * uv_sz;
   69|       |
   70|  54.9k|    uint8_t *const buf = dav1d_mem_pool_pop(cookie, pic_size + DAV1D_PICTURE_ALIGNMENT);
  ------------------
  |  |   44|  54.9k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   71|  54.9k|    if (!buf) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (71:9): [True: 0, False: 54.9k]
  ------------------
   72|  54.9k|    p->allocator_data = buf;
   73|  54.9k|    p->data[0] = buf;
   74|  54.9k|    p->data[1] = has_chroma ? buf + y_sz : NULL;
  ------------------
  |  Branch (74:18): [True: 29.4k, False: 25.5k]
  ------------------
   75|  54.9k|    p->data[2] = has_chroma ? buf + y_sz + uv_sz : NULL;
  ------------------
  |  Branch (75:18): [True: 29.4k, False: 25.5k]
  ------------------
   76|       |
   77|  54.9k|    return 0;
   78|  54.9k|}
dav1d_default_picture_release:
   80|  54.9k|void dav1d_default_picture_release(Dav1dPicture *const p, void *const cookie) {
   81|  54.9k|    dav1d_mem_pool_push(cookie, p->allocator_data);
   82|  54.9k|}
dav1d_picture_free_itut_t35:
   99|    297|void dav1d_picture_free_itut_t35(const uint8_t *const data, void *const user_data) {
  100|    297|    struct itut_t35_ctx_context *itut_t35_ctx = user_data;
  101|       |
  102|    964|    for (size_t i = 0; i < itut_t35_ctx->n_itut_t35; i++)
  ------------------
  |  Branch (102:24): [True: 667, False: 297]
  ------------------
  103|    667|        dav1d_free(itut_t35_ctx->itut_t35[i].payload);
  ------------------
  |  |  135|    667|#define dav1d_free(ptr) free(ptr)
  ------------------
  104|    297|    dav1d_free(itut_t35_ctx->itut_t35);
  ------------------
  |  |  135|    297|#define dav1d_free(ptr) free(ptr)
  ------------------
  105|    297|    dav1d_free(itut_t35_ctx);
  ------------------
  |  |  135|    297|#define dav1d_free(ptr) free(ptr)
  ------------------
  106|    297|}
dav1d_picture_copy_props:
  164|  55.3k|{
  165|  55.3k|    dav1d_data_props_copy(&p->m, props);
  166|       |
  167|  55.3k|    dav1d_ref_dec(&p->content_light_ref);
  168|  55.3k|    p->content_light_ref = content_light_ref;
  169|  55.3k|    p->content_light = content_light;
  170|  55.3k|    if (content_light_ref) dav1d_ref_inc(content_light_ref);
  ------------------
  |  Branch (170:9): [True: 1.56k, False: 53.7k]
  ------------------
  171|       |
  172|  55.3k|    dav1d_ref_dec(&p->mastering_display_ref);
  173|  55.3k|    p->mastering_display_ref = mastering_display_ref;
  174|  55.3k|    p->mastering_display = mastering_display;
  175|  55.3k|    if (mastering_display_ref) dav1d_ref_inc(mastering_display_ref);
  ------------------
  |  Branch (175:9): [True: 1.64k, False: 53.6k]
  ------------------
  176|       |
  177|  55.3k|    dav1d_ref_dec(&p->itut_t35_ref);
  178|  55.3k|    p->itut_t35_ref = itut_t35_ref;
  179|  55.3k|    p->itut_t35 = itut_t35;
  180|  55.3k|    p->n_itut_t35 = n_itut_t35;
  181|  55.3k|    if (itut_t35_ref) dav1d_ref_inc(itut_t35_ref);
  ------------------
  |  Branch (181:9): [True: 534, False: 54.7k]
  ------------------
  182|  55.3k|}
dav1d_thread_picture_alloc:
  186|  45.8k|{
  187|  45.8k|    Dav1dThreadPicture *const p = &f->sr_cur;
  188|       |
  189|  45.8k|    const int res = picture_alloc(c, &p->p, f->frame_hdr->width[1], f->frame_hdr->height,
  190|  45.8k|                                  f->seq_hdr, f->seq_hdr_ref,
  191|  45.8k|                                  f->frame_hdr, f->frame_hdr_ref,
  192|  45.8k|                                  bpc, &f->tile[0].data.m, &c->allocator,
  193|  45.8k|                                  (void **) &p->progress);
  194|  45.8k|    if (res) return res;
  ------------------
  |  Branch (194:9): [True: 0, False: 45.8k]
  ------------------
  195|       |
  196|       |    // Don't clear these flags from c->frame_flags if the frame is not going to be output.
  197|       |    // This way they will be added to the next visible frame too.
  198|  45.8k|    const int flags_mask = ((f->frame_hdr->show_frame || c->output_invisible_frames) &&
  ------------------
  |  Branch (198:30): [True: 41.3k, False: 4.47k]
  |  Branch (198:58): [True: 0, False: 4.47k]
  ------------------
  199|  41.3k|                            c->max_spatial_id == f->frame_hdr->spatial_id)
  ------------------
  |  Branch (199:29): [True: 36.2k, False: 5.07k]
  ------------------
  200|  45.8k|                           ? 0 : (PICTURE_FLAG_NEW_SEQUENCE | PICTURE_FLAG_NEW_OP_PARAMS_INFO);
  201|  45.8k|    p->flags = c->frame_flags;
  202|  45.8k|    c->frame_flags &= flags_mask;
  203|       |
  204|  45.8k|    p->visible = f->frame_hdr->show_frame;
  205|  45.8k|    p->showable = f->frame_hdr->showable_frame;
  206|       |
  207|  45.8k|    if (p->visible) {
  ------------------
  |  Branch (207:9): [True: 41.3k, False: 4.47k]
  ------------------
  208|       |        // Only add HDR10+ and T35 metadata when show frame flag is enabled
  209|  41.3k|        dav1d_picture_copy_props(&p->p, c->content_light, c->content_light_ref,
  210|  41.3k|                                 c->mastering_display, c->mastering_display_ref,
  211|  41.3k|                                 c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
  212|  41.3k|                                 &f->tile[0].data.m);
  213|       |
  214|       |        // Must be removed from the context after being attached to the frame
  215|  41.3k|        dav1d_ref_dec(&c->itut_t35_ref);
  216|  41.3k|        c->itut_t35 = NULL;
  217|  41.3k|        c->n_itut_t35 = 0;
  218|  41.3k|    } else {
  219|  4.47k|        dav1d_data_props_copy(&p->p.m, &f->tile[0].data.m);
  220|  4.47k|    }
  221|       |
  222|  45.8k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (222:9): [True: 0, False: 45.8k]
  ------------------
  223|      0|        atomic_init(&p->progress[0], 0);
  224|       |        atomic_init(&p->progress[1], 0);
  225|      0|    }
  226|  45.8k|    return res;
  227|  45.8k|}
dav1d_picture_alloc_copy:
  231|  9.14k|{
  232|  9.14k|    struct pic_ctx_context *const pic_ctx = (struct pic_ctx_context*)src->ref->const_data;
  233|  9.14k|    const int res = picture_alloc(c, dst, w, src->p.h,
  234|  9.14k|                                  src->seq_hdr, src->seq_hdr_ref,
  235|  9.14k|                                  src->frame_hdr, src->frame_hdr_ref,
  236|  9.14k|                                  src->p.bpc, &src->m, &pic_ctx->allocator,
  237|  9.14k|                                  NULL);
  238|  9.14k|    if (res) return res;
  ------------------
  |  Branch (238:9): [True: 0, False: 9.14k]
  ------------------
  239|       |
  240|  9.14k|    dav1d_picture_copy_props(dst, src->content_light, src->content_light_ref,
  241|  9.14k|                             src->mastering_display, src->mastering_display_ref,
  242|  9.14k|                             src->itut_t35, src->itut_t35_ref, src->n_itut_t35,
  243|  9.14k|                             &src->m);
  244|       |
  245|  9.14k|    return 0;
  246|  9.14k|}
dav1d_picture_ref:
  248|   515k|void dav1d_picture_ref(Dav1dPicture *const dst, const Dav1dPicture *const src) {
  249|   515k|    assert(dst != NULL);
  ------------------
  |  Branch (249:5): [True: 515k, False: 0]
  ------------------
  250|   515k|    assert(dst->data[0] == NULL);
  ------------------
  |  Branch (250:5): [True: 515k, False: 0]
  ------------------
  251|   515k|    assert(src != NULL);
  ------------------
  |  Branch (251:5): [True: 515k, False: 0]
  ------------------
  252|       |
  253|   515k|    if (src->ref) {
  ------------------
  |  Branch (253:9): [True: 515k, False: 0]
  ------------------
  254|   515k|        assert(src->data[0] != NULL);
  ------------------
  |  Branch (254:9): [True: 515k, False: 0]
  ------------------
  255|   515k|        dav1d_ref_inc(src->ref);
  256|   515k|    }
  257|   515k|    if (src->frame_hdr_ref) dav1d_ref_inc(src->frame_hdr_ref);
  ------------------
  |  Branch (257:9): [True: 515k, False: 0]
  ------------------
  258|   515k|    if (src->seq_hdr_ref) dav1d_ref_inc(src->seq_hdr_ref);
  ------------------
  |  Branch (258:9): [True: 515k, False: 0]
  ------------------
  259|   515k|    if (src->m.user_data.ref) dav1d_ref_inc(src->m.user_data.ref);
  ------------------
  |  Branch (259:9): [True: 0, False: 515k]
  ------------------
  260|   515k|    if (src->content_light_ref) dav1d_ref_inc(src->content_light_ref);
  ------------------
  |  Branch (260:9): [True: 7.47k, False: 508k]
  ------------------
  261|   515k|    if (src->mastering_display_ref) dav1d_ref_inc(src->mastering_display_ref);
  ------------------
  |  Branch (261:9): [True: 7.88k, False: 507k]
  ------------------
  262|   515k|    if (src->itut_t35_ref) dav1d_ref_inc(src->itut_t35_ref);
  ------------------
  |  Branch (262:9): [True: 1.94k, False: 513k]
  ------------------
  263|   515k|    *dst = *src;
  264|   515k|}
dav1d_picture_move_ref:
  266|  17.9k|void dav1d_picture_move_ref(Dav1dPicture *const dst, Dav1dPicture *const src) {
  267|  17.9k|    assert(dst != NULL);
  ------------------
  |  Branch (267:5): [True: 17.9k, False: 0]
  ------------------
  268|  17.9k|    assert(dst->data[0] == NULL);
  ------------------
  |  Branch (268:5): [True: 17.9k, False: 0]
  ------------------
  269|  17.9k|    assert(src != NULL);
  ------------------
  |  Branch (269:5): [True: 17.9k, False: 0]
  ------------------
  270|       |
  271|  17.9k|    if (src->ref)
  ------------------
  |  Branch (271:9): [True: 17.9k, False: 0]
  ------------------
  272|  17.9k|        assert(src->data[0] != NULL);
  ------------------
  |  Branch (272:9): [True: 17.9k, False: 0]
  ------------------
  273|       |
  274|  17.9k|    *dst = *src;
  275|  17.9k|    memset(src, 0, sizeof(*src));
  276|  17.9k|}
dav1d_thread_picture_ref:
  280|   474k|{
  281|   474k|    dav1d_picture_ref(&dst->p, &src->p);
  282|   474k|    dst->visible = src->visible;
  283|   474k|    dst->showable = src->showable;
  284|   474k|    dst->progress = src->progress;
  285|   474k|    dst->flags = src->flags;
  286|   474k|}
dav1d_picture_unref_internal:
  299|   675k|void dav1d_picture_unref_internal(Dav1dPicture *const p) {
  300|   675k|    validate_input(p != NULL);
  ------------------
  |  |   59|   675k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|   675k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 675k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  301|       |
  302|   675k|    if (p->ref) {
  ------------------
  |  Branch (302:9): [True: 570k, False: 104k]
  ------------------
  303|   570k|        validate_input(p->data[0] != NULL);
  ------------------
  |  |   59|   570k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|   570k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 570k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  304|   570k|        dav1d_ref_dec(&p->ref);
  305|   570k|    }
  306|   675k|    dav1d_ref_dec(&p->seq_hdr_ref);
  307|   675k|    dav1d_ref_dec(&p->frame_hdr_ref);
  308|   675k|    dav1d_ref_dec(&p->m.user_data.ref);
  309|   675k|    dav1d_ref_dec(&p->content_light_ref);
  310|   675k|    dav1d_ref_dec(&p->mastering_display_ref);
  311|   675k|    dav1d_ref_dec(&p->itut_t35_ref);
  312|   675k|    memset(p, 0, sizeof(*p));
  313|   675k|    dav1d_data_props_set_defaults(&p->m);
  314|   675k|}
dav1d_thread_picture_unref:
  316|   579k|void dav1d_thread_picture_unref(Dav1dThreadPicture *const p) {
  317|   579k|    dav1d_picture_unref_internal(&p->p);
  318|       |
  319|       |    p->progress = NULL;
  320|   579k|}
dav1d_picture_get_event_flags:
  322|  46.1k|enum Dav1dEventFlags dav1d_picture_get_event_flags(const Dav1dThreadPicture *const p) {
  323|  46.1k|    if (!p->flags)
  ------------------
  |  Branch (323:9): [True: 25.7k, False: 20.4k]
  ------------------
  324|  25.7k|        return 0;
  325|       |
  326|  20.4k|    enum Dav1dEventFlags flags = 0;
  327|  20.4k|    if (p->flags & PICTURE_FLAG_NEW_SEQUENCE)
  ------------------
  |  Branch (327:9): [True: 18.6k, False: 1.85k]
  ------------------
  328|  18.6k|       flags |= DAV1D_EVENT_FLAG_NEW_SEQUENCE;
  329|  20.4k|    if (p->flags & PICTURE_FLAG_NEW_OP_PARAMS_INFO)
  ------------------
  |  Branch (329:9): [True: 195, False: 20.2k]
  ------------------
  330|    195|       flags |= DAV1D_EVENT_FLAG_NEW_OP_PARAMS_INFO;
  331|       |
  332|  20.4k|    return flags;
  333|  46.1k|}
picture.c:picture_alloc:
  117|  54.9k|{
  118|  54.9k|    if (p->data[0]) {
  ------------------
  |  Branch (118:9): [True: 0, False: 54.9k]
  ------------------
  119|      0|        dav1d_log(c, "Picture already allocated!\n");
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
  120|      0|        return -1;
  121|      0|    }
  122|  54.9k|    assert(bpc > 0 && bpc <= 16);
  ------------------
  |  Branch (122:5): [True: 54.9k, False: 0]
  |  Branch (122:5): [True: 54.9k, False: 0]
  ------------------
  123|       |
  124|  54.9k|    size_t extra = c->n_fc > 1 ? sizeof(atomic_int) * 2 : 0;
  ------------------
  |  Branch (124:20): [True: 0, False: 54.9k]
  ------------------
  125|  54.9k|    struct pic_ctx_context *pic_ctx = dav1d_mem_pool_pop(c->pic_ctx_pool, extra +
  126|  54.9k|                                                         sizeof(struct pic_ctx_context));
  127|  54.9k|    if (!pic_ctx)
  ------------------
  |  Branch (127:9): [True: 0, False: 54.9k]
  ------------------
  128|      0|        return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  129|       |
  130|  54.9k|    p->p.w = w;
  131|  54.9k|    p->p.h = h;
  132|  54.9k|    p->seq_hdr = seq_hdr;
  133|  54.9k|    p->frame_hdr = frame_hdr;
  134|  54.9k|    p->p.layout = seq_hdr->layout;
  135|  54.9k|    p->p.bpc = bpc;
  136|  54.9k|    dav1d_data_props_set_defaults(&p->m);
  137|  54.9k|    const int res = p_allocator->alloc_picture_callback(p, p_allocator->cookie);
  138|  54.9k|    if (res < 0) {
  ------------------
  |  Branch (138:9): [True: 0, False: 54.9k]
  ------------------
  139|      0|        dav1d_mem_pool_push(c->pic_ctx_pool, pic_ctx);
  140|      0|        return res;
  141|      0|    }
  142|       |
  143|  54.9k|    pic_ctx->allocator = *p_allocator;
  144|  54.9k|    pic_ctx->pic = *p;
  145|  54.9k|    p->ref = dav1d_ref_init(&pic_ctx->ref, pic_ctx, free_buffer, c->pic_ctx_pool, 0);
  146|       |
  147|  54.9k|    p->seq_hdr_ref = seq_hdr_ref;
  148|  54.9k|    if (seq_hdr_ref) dav1d_ref_inc(seq_hdr_ref);
  ------------------
  |  Branch (148:9): [True: 54.9k, False: 0]
  ------------------
  149|       |
  150|  54.9k|    p->frame_hdr_ref = frame_hdr_ref;
  151|  54.9k|    if (frame_hdr_ref) dav1d_ref_inc(frame_hdr_ref);
  ------------------
  |  Branch (151:9): [True: 54.9k, False: 0]
  ------------------
  152|       |
  153|  54.9k|    if (extra && extra_ptr)
  ------------------
  |  Branch (153:9): [True: 0, False: 54.9k]
  |  Branch (153:18): [True: 0, False: 0]
  ------------------
  154|      0|        *extra_ptr = &pic_ctx->extra_data;
  155|       |
  156|  54.9k|    return 0;
  157|  54.9k|}
picture.c:free_buffer:
   91|  54.9k|static void free_buffer(const uint8_t *const data, void *const user_data) {
   92|  54.9k|    struct pic_ctx_context *pic_ctx = (struct pic_ctx_context*)data;
   93|       |
   94|  54.9k|    pic_ctx->allocator.release_picture_callback(&pic_ctx->pic,
   95|  54.9k|                                                pic_ctx->allocator.cookie);
   96|  54.9k|    dav1d_mem_pool_push(user_data, pic_ctx);
   97|  54.9k|}

dav1d_init_qm_tables:
 4684|      1|COLD void dav1d_init_qm_tables(void) {
 4685|       |    // This function is guaranteed to be called only once
 4686|       |
 4687|     16|    for (int i = 0; i < 15; i++)
  ------------------
  |  Branch (4687:21): [True: 15, False: 1]
  ------------------
 4688|     45|        for (int j = 0; j < 2; j++) {
  ------------------
  |  Branch (4688:25): [True: 30, False: 15]
  ------------------
 4689|       |            // note that the w/h in the assignment is inverted, this is on purpose
 4690|       |            // because we store coefficients transposed
 4691|     30|            dav1d_qm_tbl[i][j][RTX_4X8  ] = qm_tbl_8x4[i][j];
 4692|     30|            dav1d_qm_tbl[i][j][RTX_8X4  ] = qm_tbl_4x8[i][j];
 4693|     30|            dav1d_qm_tbl[i][j][RTX_4X16 ] = qm_tbl_16x4[i][j];
 4694|     30|            dav1d_qm_tbl[i][j][RTX_16X4 ] = qm_tbl_4x16[i][j];
 4695|     30|            dav1d_qm_tbl[i][j][RTX_8X16 ] = qm_tbl_16x8[i][j];
 4696|     30|            dav1d_qm_tbl[i][j][RTX_16X8 ] = qm_tbl_8x16[i][j];
 4697|     30|            dav1d_qm_tbl[i][j][RTX_8X32 ] = qm_tbl_32x8[i][j];
 4698|     30|            dav1d_qm_tbl[i][j][RTX_32X8 ] = qm_tbl_8x32[i][j];
 4699|     30|            dav1d_qm_tbl[i][j][RTX_16X32] = qm_tbl_32x16[i][j];
 4700|     30|            dav1d_qm_tbl[i][j][RTX_32X16] = qm_tbl_16x32[i][j];
 4701|       |
 4702|     30|            dav1d_qm_tbl[i][j][ TX_4X4  ] = qm_tbl_4x4[i][j];
 4703|     30|            dav1d_qm_tbl[i][j][ TX_8X8  ] = qm_tbl_8x8[i][j];
 4704|     30|            dav1d_qm_tbl[i][j][ TX_16X16] = qm_tbl_16x16[i][j];
 4705|     30|            dav1d_qm_tbl[i][j][ TX_32X32] = qm_tbl_32x32[i][j];
 4706|       |
 4707|     30|            dav1d_qm_tbl[i][j][ TX_64X64] = dav1d_qm_tbl[i][j][ TX_32X32];
 4708|     30|            dav1d_qm_tbl[i][j][RTX_64X32] = dav1d_qm_tbl[i][j][ TX_32X32];
 4709|     30|            dav1d_qm_tbl[i][j][RTX_64X16] = dav1d_qm_tbl[i][j][RTX_32X16];
 4710|     30|            dav1d_qm_tbl[i][j][RTX_32X64] = dav1d_qm_tbl[i][j][ TX_32X32];
 4711|     30|            dav1d_qm_tbl[i][j][RTX_16X64] = dav1d_qm_tbl[i][j][RTX_16X32];
 4712|     30|        }
 4713|       |
 4714|       |    // dav1d_qm_tbl[15][*][*] == NULL
 4715|      1|}

dav1d_recon_b_intra_8bpc:
 1179|   882k|{
 1180|   882k|    Dav1dTileState *const ts = t->ts;
 1181|   882k|    const Dav1dFrameContext *const f = t->f;
 1182|   882k|    const Dav1dDSPContext *const dsp = f->dsp;
 1183|   882k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1184|   882k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1185|   882k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1186|   882k|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1187|   882k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1188|   882k|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1189|   882k|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1190|   882k|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1191|   882k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1191:28): [True: 735k, False: 147k]
  ------------------
 1192|   735k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1192:29): [True: 688k, False: 47.3k]
  |  Branch (1192:45): [True: 23.5k, False: 23.8k]
  ------------------
 1193|   711k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1193:29): [True: 678k, False: 33.6k]
  |  Branch (1193:45): [True: 16.7k, False: 16.9k]
  ------------------
 1194|   882k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|   882k|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
 1196|       |
 1197|       |    // coefficient coding
 1198|   882k|    pixel *const edge = bitfn(t->scratch.edge) + 128;
  ------------------
  |  |   51|   882k|#define bitfn(x) x##_8bpc
  ------------------
 1199|   882k|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
 1200|       |
 1201|   882k|    const int intra_edge_filter_flag = f->seq_hdr->intra_edge_filter << 10;
 1202|       |
 1203|  1.78M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (1203:26): [True: 906k, False: 882k]
  ------------------
 1204|   906k|        const int sub_h4 = imin(h4, 16 + init_y);
 1205|   906k|        const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
 1206|  1.85M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (1206:30): [True: 952k, False: 906k]
  ------------------
 1207|   952k|            if (b->pal_sz[0]) {
  ------------------
  |  Branch (1207:17): [True: 21.0k, False: 931k]
  ------------------
 1208|  21.0k|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1209|  21.0k|                             4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
  ------------------
  |  |   53|  21.0k|#define PXSTRIDE(x) (x)
  ------------------
 1210|  21.0k|                const uint8_t *pal_idx;
 1211|  21.0k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1211:21): [True: 0, False: 21.0k]
  ------------------
 1212|      0|                    const int p = t->frame_thread.pass & 1;
 1213|      0|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1213:21): [True: 0, False: 0]
  ------------------
 1214|      0|                    pal_idx = ts->frame_thread[p].pal_idx;
 1215|      0|                    ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1216|  21.0k|                } else {
 1217|  21.0k|                    pal_idx = t->scratch.pal_idx_y;
 1218|  21.0k|                }
 1219|  21.0k|                const pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (1219:42): [True: 0, False: 21.0k]
  ------------------
 1220|      0|                    f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1221|      0|                                        ((t->bx >> 1) + (t->by & 1))][0] :
 1222|  21.0k|                    bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|  21.0k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  21.0k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1223|  21.0k|                f->dsp->ipred.pal_pred(dst, f->cur.stride[0], pal,
 1224|  21.0k|                                       pal_idx, bw4 * 4, bh4 * 4);
 1225|  21.0k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  21.0k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 21.0k]
  |  |  ------------------
  |  |   35|  21.0k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  21.0k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1226|      0|                    hex_dump(dst, PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1227|      0|                             bw4 * 4, bh4 * 4, "y-pal-pred");
 1228|  21.0k|            }
 1229|       |
 1230|   952k|            const int intra_flags = (sm_flag(t->a, bx4) |
 1231|   952k|                                     sm_flag(&t->l, by4) |
 1232|   952k|                                     intra_edge_filter_flag);
 1233|   952k|            const int sb_has_tr = init_x + 16 < w4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1233:35): [True: 46.3k, False: 906k]
  |  Branch (1233:58): [True: 23.7k, False: 882k]
  ------------------
 1234|   906k|                              intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT;
 1235|   952k|            const int sb_has_bl = init_x ? 0 : init_y + 16 < h4 ? 1 :
  ------------------
  |  Branch (1235:35): [True: 46.3k, False: 906k]
  |  Branch (1235:48): [True: 23.7k, False: 882k]
  ------------------
 1236|   906k|                              intra_edge_flags & EDGE_I444_LEFT_HAS_BOTTOM;
 1237|   952k|            int y, x;
 1238|   952k|            const int sub_w4 = imin(w4, init_x + 16);
 1239|  2.32M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (1239:47): [True: 1.37M, False: 952k]
  ------------------
 1240|  1.37M|                 y += t_dim->h, t->by += t_dim->h)
 1241|  1.37M|            {
 1242|  1.37M|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1243|  1.37M|                               4 * (t->by * PXSTRIDE(f->cur.stride[0]) +
  ------------------
  |  |   53|  1.37M|#define PXSTRIDE(x) (x)
  ------------------
 1244|  1.37M|                                    t->bx + init_x);
 1245|  4.86M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (1245:51): [True: 3.49M, False: 1.37M]
  ------------------
 1246|  3.49M|                     x += t_dim->w, t->bx += t_dim->w)
 1247|  3.49M|                {
 1248|  3.49M|                    if (b->pal_sz[0]) goto skip_y_pred;
  ------------------
  |  Branch (1248:25): [True: 36.6k, False: 3.46M]
  ------------------
 1249|       |
 1250|  3.46M|                    int angle = b->y_angle;
 1251|  3.46M|                    const enum EdgeFlags edge_flags =
 1252|  3.46M|                        (((y > init_y || !sb_has_tr) && (x + t_dim->w >= sub_w4)) ?
  ------------------
  |  Branch (1252:28): [True: 2.12M, False: 1.34M]
  |  Branch (1252:42): [True: 446k, False: 893k]
  |  Branch (1252:57): [True: 719k, False: 1.84M]
  ------------------
 1253|  2.74M|                             0 : EDGE_I444_TOP_HAS_RIGHT) |
 1254|  3.46M|                        ((x > init_x || (!sb_has_bl && y + t_dim->h >= sub_h4)) ?
  ------------------
  |  Branch (1254:27): [True: 2.11M, False: 1.34M]
  |  Branch (1254:42): [True: 859k, False: 486k]
  |  Branch (1254:56): [True: 552k, False: 307k]
  ------------------
 1255|  2.66M|                             0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1256|  3.46M|                    const pixel *top_sb_edge = NULL;
 1257|  3.46M|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1257:25): [True: 544k, False: 2.91M]
  ------------------
 1258|   544k|                        top_sb_edge = f->ipred_edge[0];
 1259|   544k|                        const int sby = t->by >> f->sb_shift;
 1260|   544k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1261|   544k|                    }
 1262|  3.46M|                    const enum IntraPredMode m =
 1263|  3.46M|                        bytefn(dav1d_prepare_intra_edges)(t->bx,
  ------------------
  |  |   87|  3.46M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  3.46M|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1264|  3.46M|                                                          t->bx > ts->tiling.col_start,
 1265|  3.46M|                                                          t->by,
 1266|  3.46M|                                                          t->by > ts->tiling.row_start,
 1267|  3.46M|                                                          ts->tiling.col_end,
 1268|  3.46M|                                                          ts->tiling.row_end,
 1269|  3.46M|                                                          edge_flags, dst,
 1270|  3.46M|                                                          f->cur.stride[0], top_sb_edge,
 1271|  3.46M|                                                          b->y_mode, &angle,
 1272|  3.46M|                                                          t_dim->w, t_dim->h,
 1273|  3.46M|                                                          f->seq_hdr->intra_edge_filter,
 1274|  3.46M|                                                          edge HIGHBD_CALL_SUFFIX);
 1275|  3.46M|                    dsp->ipred.intra_pred[m](dst, f->cur.stride[0], edge,
 1276|  3.46M|                                             t_dim->w * 4, t_dim->h * 4,
 1277|  3.46M|                                             angle | intra_flags,
 1278|  3.46M|                                             4 * f->bw - 4 * t->bx,
 1279|  3.46M|                                             4 * f->bh - 4 * t->by
 1280|  3.46M|                                             HIGHBD_CALL_SUFFIX);
 1281|       |
 1282|  3.46M|                    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  3.46M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 3.46M]
  |  |  ------------------
  |  |   35|  3.46M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  3.46M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1283|      0|                        hex_dump(edge - t_dim->h * 4, t_dim->h * 4,
 1284|      0|                                 t_dim->h * 4, 2, "l");
 1285|      0|                        hex_dump(edge, 0, 1, 1, "tl");
 1286|      0|                        hex_dump(edge + 1, t_dim->w * 4,
 1287|      0|                                 t_dim->w * 4, 2, "t");
 1288|      0|                        hex_dump(dst, f->cur.stride[0],
 1289|      0|                                 t_dim->w * 4, t_dim->h * 4, "y-intra-pred");
 1290|      0|                    }
 1291|       |
 1292|  3.49M|                skip_y_pred: {}
 1293|  3.49M|                    if (!b->skip) {
  ------------------
  |  Branch (1293:25): [True: 1.51M, False: 1.98M]
  ------------------
 1294|  1.51M|                        coef *cf;
 1295|  1.51M|                        int eob;
 1296|  1.51M|                        enum TxfmType txtp;
 1297|  1.51M|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1297:29): [True: 0, False: 1.51M]
  ------------------
 1298|      0|                            const int p = t->frame_thread.pass & 1;
 1299|      0|                            const int cbi = *ts->frame_thread[p].cbi++;
 1300|      0|                            cf = ts->frame_thread[p].cf;
 1301|      0|                            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
 1302|      0|                            eob  = cbi >> 5;
 1303|      0|                            txtp = cbi & 0x1f;
 1304|  1.51M|                        } else {
 1305|  1.51M|                            uint8_t cf_ctx;
 1306|  1.51M|                            cf = bitfn(t->cf);
  ------------------
  |  |   51|  1.51M|#define bitfn(x) x##_8bpc
  ------------------
 1307|  1.51M|                            eob = decode_coefs(t, &t->a->lcoef[bx4 + x],
 1308|  1.51M|                                               &t->l.lcoef[by4 + y], b->tx, bs,
 1309|  1.51M|                                               b, 1, 0, cf, &txtp, &cf_ctx);
 1310|  1.51M|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.51M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.51M]
  |  |  ------------------
  |  |   35|  1.51M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.51M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1311|      0|                                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
 1312|      0|                                       b->tx, txtp, eob, ts->msac.rng);
 1313|  1.51M|                            dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
 1314|  1.51M|                            dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
 1315|  1.51M|                        }
 1316|  1.51M|                        if (eob >= 0) {
  ------------------
  |  Branch (1316:29): [True: 889k, False: 622k]
  ------------------
 1317|   889k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   889k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 889k]
  |  |  ------------------
  |  |   35|   889k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   889k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1318|      0|                                coef_dump(cf, imin(t_dim->h, 8) * 4,
 1319|      0|                                          imin(t_dim->w, 8) * 4, 3, "dq");
 1320|   889k|                            dsp->itx.itxfm_add[b->tx]
 1321|   889k|                                              [txtp](dst,
 1322|   889k|                                                     f->cur.stride[0],
 1323|   889k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
 1324|   889k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   889k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 889k]
  |  |  ------------------
  |  |   35|   889k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   889k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1325|      0|                                hex_dump(dst, f->cur.stride[0],
 1326|      0|                                         t_dim->w * 4, t_dim->h * 4, "recon");
 1327|   889k|                        }
 1328|  1.98M|                    } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1328:32): [True: 1.98M, False: 0]
  ------------------
 1329|  1.98M|                        dav1d_memset_pow2[t_dim->lw](&t->a->lcoef[bx4 + x], 0x40);
 1330|  1.98M|                        dav1d_memset_pow2[t_dim->lh](&t->l.lcoef[by4 + y], 0x40);
 1331|  1.98M|                    }
 1332|  3.49M|                    dst += 4 * t_dim->w;
 1333|  3.49M|                }
 1334|  1.37M|                t->bx -= x;
 1335|  1.37M|            }
 1336|   952k|            t->by -= y;
 1337|       |
 1338|   952k|            if (!has_chroma) continue;
  ------------------
  |  Branch (1338:17): [True: 209k, False: 743k]
  ------------------
 1339|       |
 1340|   743k|            const ptrdiff_t stride = f->cur.stride[1];
 1341|       |
 1342|   743k|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1342:17): [True: 138k, False: 605k]
  ------------------
 1343|   138k|                assert(!init_x && !init_y);
  ------------------
  |  Branch (1343:17): [True: 138k, False: 0]
  |  Branch (1343:17): [True: 138k, False: 0]
  ------------------
 1344|       |
 1345|   138k|                int16_t *const ac = t->scratch.ac;
 1346|   138k|                pixel *y_src = ((pixel *) f->cur.data[0]) + 4 * (t->bx & ~ss_hor) +
 1347|   138k|                                 4 * (t->by & ~ss_ver) * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|   138k|#define PXSTRIDE(x) (x)
  ------------------
 1348|   138k|                const ptrdiff_t uv_off = 4 * ((t->bx >> ss_hor) +
 1349|   138k|                                              (t->by >> ss_ver) * PXSTRIDE(stride));
  ------------------
  |  |   53|   138k|#define PXSTRIDE(x) (x)
  ------------------
 1350|   138k|                pixel *const uv_dst[2] = { ((pixel *) f->cur.data[1]) + uv_off,
 1351|   138k|                                           ((pixel *) f->cur.data[2]) + uv_off };
 1352|       |
 1353|   138k|                const int furthest_r =
 1354|   138k|                    ((cw4 << ss_hor) + t_dim->w - 1) & ~(t_dim->w - 1);
 1355|   138k|                const int furthest_b =
 1356|   138k|                    ((ch4 << ss_ver) + t_dim->h - 1) & ~(t_dim->h - 1);
 1357|   138k|                dsp->ipred.cfl_ac[f->cur.p.layout - 1](ac, y_src, f->cur.stride[0],
 1358|   138k|                                                         cbw4 - (furthest_r >> ss_hor),
 1359|   138k|                                                         cbh4 - (furthest_b >> ss_ver),
 1360|   138k|                                                         cbw4 * 4, cbh4 * 4);
 1361|   415k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1361:34): [True: 277k, False: 138k]
  ------------------
 1362|   277k|                    if (!b->cfl_alpha[pl]) continue;
  ------------------
  |  Branch (1362:25): [True: 47.4k, False: 229k]
  ------------------
 1363|   229k|                    int angle = 0;
 1364|   229k|                    const pixel *top_sb_edge = NULL;
 1365|   229k|                    if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1365:25): [True: 74.2k, False: 155k]
  ------------------
 1366|  74.2k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1367|  74.2k|                        const int sby = t->by >> f->sb_shift;
 1368|  74.2k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1369|  74.2k|                    }
 1370|   229k|                    const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1371|   229k|                    const int xstart = ts->tiling.col_start >> ss_hor;
 1372|   229k|                    const int ystart = ts->tiling.row_start >> ss_ver;
 1373|   229k|                    const enum IntraPredMode m =
 1374|   229k|                        bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|   229k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   229k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1375|   229k|                                                          ypos, ypos > ystart,
 1376|   229k|                                                          ts->tiling.col_end >> ss_hor,
 1377|   229k|                                                          ts->tiling.row_end >> ss_ver,
 1378|   229k|                                                          0, uv_dst[pl], stride,
 1379|   229k|                                                          top_sb_edge, DC_PRED, &angle,
 1380|   229k|                                                          uv_t_dim->w, uv_t_dim->h, 0,
 1381|   229k|                                                          edge HIGHBD_CALL_SUFFIX);
 1382|   229k|                    dsp->ipred.cfl_pred[m](uv_dst[pl], stride, edge,
 1383|   229k|                                           uv_t_dim->w * 4,
 1384|   229k|                                           uv_t_dim->h * 4,
 1385|   229k|                                           ac, b->cfl_alpha[pl]
 1386|   229k|                                           HIGHBD_CALL_SUFFIX);
 1387|   229k|                }
 1388|   138k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|   138k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 138k]
  |  |  ------------------
  |  |   35|   138k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   138k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1389|      0|                    ac_dump(ac, 4*cbw4, 4*cbh4, "ac");
 1390|      0|                    hex_dump(uv_dst[0], stride, cbw4 * 4, cbh4 * 4, "u-cfl-pred");
 1391|      0|                    hex_dump(uv_dst[1], stride, cbw4 * 4, cbh4 * 4, "v-cfl-pred");
 1392|      0|                }
 1393|   605k|            } else if (b->pal_sz[1]) {
  ------------------
  |  Branch (1393:24): [True: 4.92k, False: 600k]
  ------------------
 1394|  4.92k|                const ptrdiff_t uv_dstoff = 4 * ((t->bx >> ss_hor) +
 1395|  4.92k|                                              (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
  ------------------
  |  |   53|  4.92k|#define PXSTRIDE(x) (x)
  ------------------
 1396|  4.92k|                const pixel (*pal)[8];
 1397|  4.92k|                const uint8_t *pal_idx;
 1398|  4.92k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1398:21): [True: 0, False: 4.92k]
  ------------------
 1399|      0|                    const int p = t->frame_thread.pass & 1;
 1400|      0|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1400:21): [True: 0, False: 0]
  ------------------
 1401|      0|                    pal = f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1402|      0|                                              ((t->bx >> 1) + (t->by & 1))];
 1403|      0|                    pal_idx = ts->frame_thread[p].pal_idx;
 1404|      0|                    ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1405|  4.92k|                } else {
 1406|  4.92k|                    pal = bytefn(t->scratch.pal);
  ------------------
  |  |   87|  4.92k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  4.92k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1407|  4.92k|                    pal_idx = t->scratch.pal_idx_uv;
 1408|  4.92k|                }
 1409|       |
 1410|  4.92k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[1]) + uv_dstoff,
 1411|  4.92k|                                       f->cur.stride[1], pal[1],
 1412|  4.92k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1413|  4.92k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[2]) + uv_dstoff,
 1414|  4.92k|                                       f->cur.stride[1], pal[2],
 1415|  4.92k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1416|  4.92k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  4.92k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 4.92k]
  |  |  ------------------
  |  |   35|  4.92k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  4.92k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1417|      0|                    hex_dump(((pixel *) f->cur.data[1]) + uv_dstoff,
 1418|      0|                             PXSTRIDE(f->cur.stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1419|      0|                             cbw4 * 4, cbh4 * 4, "u-pal-pred");
 1420|      0|                    hex_dump(((pixel *) f->cur.data[2]) + uv_dstoff,
 1421|      0|                             PXSTRIDE(f->cur.stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1422|      0|                             cbw4 * 4, cbh4 * 4, "v-pal-pred");
 1423|      0|                }
 1424|  4.92k|            }
 1425|       |
 1426|   743k|            const int sm_uv_fl = sm_uv_flag(t->a, cbx4) |
 1427|   743k|                                 sm_uv_flag(&t->l, cby4);
 1428|   743k|            const int uv_sb_has_tr =
 1429|   743k|                ((init_x + 16) >> ss_hor) < cw4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1429:17): [True: 32.3k, False: 711k]
  |  Branch (1429:55): [True: 16.5k, False: 694k]
  ------------------
 1430|   711k|                intra_edge_flags & (EDGE_I420_TOP_HAS_RIGHT >> (f->cur.p.layout - 1));
 1431|   743k|            const int uv_sb_has_bl =
 1432|   743k|                init_x ? 0 : ((init_y + 16) >> ss_ver) < ch4 ? 1 :
  ------------------
  |  Branch (1432:17): [True: 32.3k, False: 711k]
  |  Branch (1432:30): [True: 16.5k, False: 694k]
  ------------------
 1433|   711k|                intra_edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM >> (f->cur.p.layout - 1));
 1434|   743k|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
 1435|  2.23M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1435:30): [True: 1.48M, False: 743k]
  ------------------
 1436|  3.28M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (1436:61): [True: 1.79M, False: 1.48M]
  ------------------
 1437|  1.79M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
 1438|  1.79M|                {
 1439|  1.79M|                    pixel *dst = ((pixel *) f->cur.data[1 + pl]) +
 1440|  1.79M|                                   4 * ((t->by >> ss_ver) * PXSTRIDE(stride) +
  ------------------
  |  |   53|  1.79M|#define PXSTRIDE(x) (x)
  ------------------
 1441|  1.79M|                                        ((t->bx + init_x) >> ss_hor));
 1442|  4.80M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (1442:65): [True: 3.01M, False: 1.79M]
  ------------------
 1443|  3.01M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
 1444|  3.01M|                    {
 1445|  3.01M|                        if ((b->uv_mode == CFL_PRED && b->cfl_alpha[pl]) ||
  ------------------
  |  Branch (1445:30): [True: 277k, False: 2.73M]
  |  Branch (1445:56): [True: 229k, False: 47.4k]
  ------------------
 1446|  2.78M|                            b->pal_sz[1])
  ------------------
  |  Branch (1446:29): [True: 16.7k, False: 2.76M]
  ------------------
 1447|   246k|                        {
 1448|   246k|                            goto skip_uv_pred;
 1449|   246k|                        }
 1450|       |
 1451|  2.76M|                        int angle = b->uv_angle;
 1452|       |                        // this probably looks weird because we're using
 1453|       |                        // luma flags in a chroma loop, but that's because
 1454|       |                        // prepare_intra_edges() expects luma flags as input
 1455|  2.76M|                        const enum EdgeFlags edge_flags =
 1456|  2.76M|                            (((y > (init_y >> ss_ver) || !uv_sb_has_tr) &&
  ------------------
  |  Branch (1456:32): [True: 1.33M, False: 1.43M]
  |  Branch (1456:58): [True: 497k, False: 941k]
  ------------------
 1457|  1.82M|                              (x + uv_t_dim->w >= sub_cw4)) ?
  ------------------
  |  Branch (1457:31): [True: 767k, False: 1.06M]
  ------------------
 1458|  2.00M|                                 0 : EDGE_I444_TOP_HAS_RIGHT) |
 1459|  2.76M|                            ((x > (init_x >> ss_hor) ||
  ------------------
  |  Branch (1459:31): [True: 1.21M, False: 1.55M]
  ------------------
 1460|  1.55M|                              (!uv_sb_has_bl && y + uv_t_dim->h >= sub_ch4)) ?
  ------------------
  |  Branch (1460:32): [True: 944k, False: 608k]
  |  Branch (1460:49): [True: 708k, False: 236k]
  ------------------
 1461|  1.92M|                                 0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1462|  2.76M|                        const pixel *top_sb_edge = NULL;
 1463|  2.76M|                        if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1463:29): [True: 525k, False: 2.24M]
  ------------------
 1464|   525k|                            top_sb_edge = f->ipred_edge[1 + pl];
 1465|   525k|                            const int sby = t->by >> f->sb_shift;
 1466|   525k|                            top_sb_edge += f->sb128w * 128 * (sby - 1);
 1467|   525k|                        }
 1468|  2.76M|                        const enum IntraPredMode uv_mode =
 1469|  2.76M|                             b->uv_mode == CFL_PRED ? DC_PRED : b->uv_mode;
  ------------------
  |  Branch (1469:30): [True: 47.4k, False: 2.72M]
  ------------------
 1470|  2.76M|                        const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1471|  2.76M|                        const int xstart = ts->tiling.col_start >> ss_hor;
 1472|  2.76M|                        const int ystart = ts->tiling.row_start >> ss_ver;
 1473|  2.76M|                        const enum IntraPredMode m =
 1474|  2.76M|                            bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  2.76M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  2.76M|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1475|  2.76M|                                                              ypos, ypos > ystart,
 1476|  2.76M|                                                              ts->tiling.col_end >> ss_hor,
 1477|  2.76M|                                                              ts->tiling.row_end >> ss_ver,
 1478|  2.76M|                                                              edge_flags, dst, stride,
 1479|  2.76M|                                                              top_sb_edge, uv_mode,
 1480|  2.76M|                                                              &angle, uv_t_dim->w,
 1481|  2.76M|                                                              uv_t_dim->h,
 1482|  2.76M|                                                              f->seq_hdr->intra_edge_filter,
 1483|  2.76M|                                                              edge HIGHBD_CALL_SUFFIX);
 1484|  2.76M|                        angle |= intra_edge_filter_flag;
 1485|  2.76M|                        dsp->ipred.intra_pred[m](dst, stride, edge,
 1486|  2.76M|                                                 uv_t_dim->w * 4,
 1487|  2.76M|                                                 uv_t_dim->h * 4,
 1488|  2.76M|                                                 angle | sm_uv_fl,
 1489|  2.76M|                                                 (4 * f->bw + ss_hor -
 1490|  2.76M|                                                  4 * (t->bx & ~ss_hor)) >> ss_hor,
 1491|  2.76M|                                                 (4 * f->bh + ss_ver -
 1492|  2.76M|                                                  4 * (t->by & ~ss_ver)) >> ss_ver
 1493|  2.76M|                                                 HIGHBD_CALL_SUFFIX);
 1494|  2.76M|                        if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  2.76M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.76M]
  |  |  ------------------
  |  |   35|  2.76M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.76M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                      if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1495|      0|                            hex_dump(edge - uv_t_dim->h * 4, uv_t_dim->h * 4,
 1496|      0|                                     uv_t_dim->h * 4, 2, "l");
 1497|      0|                            hex_dump(edge, 0, 1, 1, "tl");
 1498|      0|                            hex_dump(edge + 1, uv_t_dim->w * 4,
 1499|      0|                                     uv_t_dim->w * 4, 2, "t");
 1500|      0|                            hex_dump(dst, stride, uv_t_dim->w * 4,
 1501|      0|                                     uv_t_dim->h * 4, pl ? "v-intra-pred" : "u-intra-pred");
  ------------------
  |  Branch (1501:55): [True: 0, False: 0]
  ------------------
 1502|      0|                        }
 1503|       |
 1504|  3.01M|                    skip_uv_pred: {}
 1505|  3.01M|                        if (!b->skip) {
  ------------------
  |  Branch (1505:29): [True: 1.39M, False: 1.61M]
  ------------------
 1506|  1.39M|                            enum TxfmType txtp;
 1507|  1.39M|                            int eob;
 1508|  1.39M|                            coef *cf;
 1509|  1.39M|                            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1509:33): [True: 0, False: 1.39M]
  ------------------
 1510|      0|                                const int p = t->frame_thread.pass & 1;
 1511|      0|                                const int cbi = *ts->frame_thread[p].cbi++;
 1512|      0|                                cf = ts->frame_thread[p].cf;
 1513|      0|                                ts->frame_thread[p].cf += uv_t_dim->w * uv_t_dim->h * 16;
 1514|      0|                                eob  = cbi >> 5;
 1515|      0|                                txtp = cbi & 0x1f;
 1516|  1.39M|                            } else {
 1517|  1.39M|                                uint8_t cf_ctx;
 1518|  1.39M|                                cf = bitfn(t->cf);
  ------------------
  |  |   51|  1.39M|#define bitfn(x) x##_8bpc
  ------------------
 1519|  1.39M|                                eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1520|  1.39M|                                                   &t->l.ccoef[pl][cby4 + y],
 1521|  1.39M|                                                   b->uvtx, bs, b, 1, 1 + pl, cf,
 1522|  1.39M|                                                   &txtp, &cf_ctx);
 1523|  1.39M|                                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.39M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.39M]
  |  |  ------------------
  |  |   35|  1.39M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.39M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1524|      0|                                    printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1525|      0|                                           "txtp=%d,eob=%d]: r=%d [x=%d,cbx4=%d]\n",
 1526|      0|                                           pl, b->uvtx, txtp, eob, ts->msac.rng, x, cbx4);
 1527|  1.39M|                                int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1528|  1.39M|                                int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1529|  1.39M|                                dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1530|  1.39M|                                dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1531|  1.39M|                            }
 1532|  1.39M|                            if (eob >= 0) {
  ------------------
  |  Branch (1532:33): [True: 356k, False: 1.04M]
  ------------------
 1533|   356k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   356k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 356k]
  |  |  ------------------
  |  |   35|   356k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   356k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1534|      0|                                    coef_dump(cf, uv_t_dim->h * 4,
 1535|      0|                                              uv_t_dim->w * 4, 3, "dq");
 1536|   356k|                                dsp->itx.itxfm_add[b->uvtx]
 1537|   356k|                                                  [txtp](dst, stride,
 1538|   356k|                                                         cf, eob HIGHBD_CALL_SUFFIX);
 1539|   356k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   356k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 356k]
  |  |  ------------------
  |  |   35|   356k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   356k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1540|      0|                                    hex_dump(dst, stride, uv_t_dim->w * 4,
 1541|      0|                                             uv_t_dim->h * 4, "recon");
 1542|   356k|                            }
 1543|  1.61M|                        } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1543:36): [True: 1.61M, False: 0]
  ------------------
 1544|  1.61M|                            dav1d_memset_pow2[uv_t_dim->lw](&t->a->ccoef[pl][cbx4 + x], 0x40);
 1545|  1.61M|                            dav1d_memset_pow2[uv_t_dim->lh](&t->l.ccoef[pl][cby4 + y], 0x40);
 1546|  1.61M|                        }
 1547|  3.01M|                        dst += uv_t_dim->w * 4;
 1548|  3.01M|                    }
 1549|  1.79M|                    t->bx -= x << ss_hor;
 1550|  1.79M|                }
 1551|  1.48M|                t->by -= y << ss_ver;
 1552|  1.48M|            }
 1553|   743k|        }
 1554|   906k|    }
 1555|   882k|}
dav1d_recon_b_inter_8bpc:
 1559|   593k|{
 1560|   593k|    Dav1dTileState *const ts = t->ts;
 1561|   593k|    const Dav1dFrameContext *const f = t->f;
 1562|   593k|    const Dav1dDSPContext *const dsp = f->dsp;
 1563|   593k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1564|   593k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1565|   593k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1566|   593k|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1567|   593k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1568|   593k|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1569|   593k|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1570|   593k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1570:28): [True: 395k, False: 197k]
  ------------------
 1571|   395k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1571:29): [True: 358k, False: 37.6k]
  |  Branch (1571:45): [True: 18.7k, False: 18.8k]
  ------------------
 1572|   376k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1572:29): [True: 352k, False: 24.3k]
  |  Branch (1572:45): [True: 12.1k, False: 12.1k]
  ------------------
 1573|   593k|    const int chr_layout_idx = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I400 ? 0 :
  ------------------
  |  Branch (1573:32): [True: 197k, False: 395k]
  ------------------
 1574|   593k|                               DAV1D_PIXEL_LAYOUT_I444 - f->cur.p.layout;
 1575|   593k|    int res;
 1576|       |
 1577|       |    // prediction
 1578|   593k|    const int cbh4 = (bh4 + ss_ver) >> ss_ver, cbw4 = (bw4 + ss_hor) >> ss_hor;
 1579|   593k|    pixel *dst = ((pixel *) f->cur.data[0]) +
 1580|   593k|        4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
  ------------------
  |  |   53|   593k|#define PXSTRIDE(x) (x)
  ------------------
 1581|   593k|    const ptrdiff_t uvdstoff =
 1582|   593k|        4 * ((t->bx >> ss_hor) + (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
  ------------------
  |  |   53|   593k|#define PXSTRIDE(x) (x)
  ------------------
 1583|   593k|    if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|   593k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|   593k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 162k, False: 430k]
  |  |  ------------------
  ------------------
 1584|       |        // intrabc
 1585|   162k|        assert(!f->frame_hdr->super_res.enabled);
  ------------------
  |  Branch (1585:9): [True: 162k, False: 0]
  ------------------
 1586|   162k|        res = mc(t, dst, NULL, f->cur.stride[0], bw4, bh4, t->bx, t->by, 0,
 1587|   162k|                 b->mv[0], &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1588|   162k|        if (res) return res;
  ------------------
  |  Branch (1588:13): [True: 0, False: 162k]
  ------------------
 1589|   404k|        if (has_chroma) for (int pl = 1; pl < 3; pl++) {
  ------------------
  |  Branch (1589:13): [True: 134k, False: 28.0k]
  |  Branch (1589:42): [True: 269k, False: 134k]
  ------------------
 1590|   269k|            res = mc(t, ((pixel *)f->cur.data[pl]) + uvdstoff, NULL, f->cur.stride[1],
 1591|   269k|                     bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1592|   269k|                     t->bx & ~ss_hor, t->by & ~ss_ver, pl, b->mv[0],
 1593|   269k|                     &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1594|   269k|            if (res) return res;
  ------------------
  |  Branch (1594:17): [True: 0, False: 269k]
  ------------------
 1595|   269k|        }
 1596|   430k|    } else if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1596:16): [True: 361k, False: 68.7k]
  ------------------
 1597|   361k|        const Dav1dThreadPicture *const refp = &f->refp[b->ref[0]];
 1598|   361k|        const enum Filter2d filter_2d = b->filter2d;
 1599|       |
 1600|   361k|        if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1600:13): [True: 243k, False: 117k]
  ------------------
 1601|   243k|            ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1601:15): [True: 85.1k, False: 158k]
  |  Branch (1601:44): [True: 6.49k, False: 78.6k]
  ------------------
 1602|   237k|             (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1602:15): [True: 32.4k, False: 205k]
  |  Branch (1602:44): [True: 28.2k, False: 4.13k]
  ------------------
 1603|  34.7k|        {
 1604|  34.7k|            res = warp_affine(t, dst, NULL, f->cur.stride[0], b_dim, 0, refp,
 1605|  34.7k|                              b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1605:31): [True: 28.2k, False: 6.49k]
  ------------------
 1606|  34.7k|                                  &f->frame_hdr->gmv[b->ref[0]]);
 1607|  34.7k|            if (res) return res;
  ------------------
  |  Branch (1607:17): [True: 0, False: 34.7k]
  ------------------
 1608|   327k|        } else {
 1609|   327k|            res = mc(t, dst, NULL, f->cur.stride[0],
 1610|   327k|                     bw4, bh4, t->bx, t->by, 0, b->mv[0], refp, b->ref[0], filter_2d);
 1611|   327k|            if (res) return res;
  ------------------
  |  Branch (1611:17): [True: 0, False: 327k]
  ------------------
 1612|   327k|            if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1612:17): [True: 70.2k, False: 256k]
  ------------------
 1613|  70.2k|                res = obmc(t, dst, f->cur.stride[0], b_dim, 0, bx4, by4, w4, h4);
 1614|  70.2k|                if (res) return res;
  ------------------
  |  Branch (1614:21): [True: 0, False: 70.2k]
  ------------------
 1615|  70.2k|            }
 1616|   327k|        }
 1617|   361k|        if (b->interintra_type) {
  ------------------
  |  Branch (1617:13): [True: 19.8k, False: 342k]
  ------------------
 1618|  19.8k|            pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   51|  19.8k|#define bitfn(x) x##_8bpc
  ------------------
 1619|  19.8k|            enum IntraPredMode m = b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1619:36): [True: 3.08k, False: 16.7k]
  ------------------
 1620|  16.7k|                                   SMOOTH_PRED : b->interintra_mode;
 1621|  19.8k|            pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   51|  19.8k|#define bitfn(x) x##_8bpc
  ------------------
 1622|  19.8k|            int angle = 0;
 1623|  19.8k|            const pixel *top_sb_edge = NULL;
 1624|  19.8k|            if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1624:17): [True: 3.13k, False: 16.7k]
  ------------------
 1625|  3.13k|                top_sb_edge = f->ipred_edge[0];
 1626|  3.13k|                const int sby = t->by >> f->sb_shift;
 1627|  3.13k|                top_sb_edge += f->sb128w * 128 * (sby - 1);
 1628|  3.13k|            }
 1629|  19.8k|            m = bytefn(dav1d_prepare_intra_edges)(t->bx, t->bx > ts->tiling.col_start,
  ------------------
  |  |   87|  19.8k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  19.8k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1630|  19.8k|                                                  t->by, t->by > ts->tiling.row_start,
 1631|  19.8k|                                                  ts->tiling.col_end, ts->tiling.row_end,
 1632|  19.8k|                                                  0, dst, f->cur.stride[0], top_sb_edge,
 1633|  19.8k|                                                  m, &angle, bw4, bh4, 0, tl_edge
 1634|  19.8k|                                                  HIGHBD_CALL_SUFFIX);
 1635|  19.8k|            dsp->ipred.intra_pred[m](tmp, 4 * bw4 * sizeof(pixel),
 1636|  19.8k|                                     tl_edge, bw4 * 4, bh4 * 4, 0, 0, 0
 1637|  19.8k|                                     HIGHBD_CALL_SUFFIX);
 1638|  19.8k|            dsp->mc.blend(dst, f->cur.stride[0], tmp,
 1639|  19.8k|                          bw4 * 4, bh4 * 4, II_MASK(0, bs, b));
  ------------------
  |  |   83|  19.8k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  19.8k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 14.8k, False: 4.96k]
  |  |  ------------------
  |  |   85|  19.8k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  19.8k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1640|  19.8k|        }
 1641|       |
 1642|   361k|        if (!has_chroma) goto skip_inter_chroma_pred;
  ------------------
  |  Branch (1642:13): [True: 171k, False: 190k]
  ------------------
 1643|       |
 1644|       |        // sub8x8 derivation
 1645|   190k|        int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1645:25): [True: 8.63k, False: 182k]
  |  Branch (1645:42): [True: 6.29k, False: 175k]
  ------------------
 1646|   190k|        refmvs_block *const *r;
 1647|   190k|        if (is_sub8x8) {
  ------------------
  |  Branch (1647:13): [True: 14.9k, False: 175k]
  ------------------
 1648|  14.9k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1648:13): [True: 14.9k, False: 0]
  ------------------
 1649|  14.9k|            r = &t->rt.r[(t->by & 31) + 5];
 1650|  14.9k|            if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1650:17): [True: 8.63k, False: 6.29k]
  ------------------
 1651|  14.9k|            if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1651:17): [True: 9.36k, False: 5.56k]
  ------------------
 1652|  14.9k|            if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1652:17): [True: 8.63k, False: 6.29k]
  |  Branch (1652:29): [True: 3.07k, False: 5.56k]
  ------------------
 1653|  3.07k|                is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1654|  14.9k|        }
 1655|       |
 1656|       |        // chroma prediction
 1657|   190k|        if (is_sub8x8) {
  ------------------
  |  Branch (1657:13): [True: 13.7k, False: 176k]
  ------------------
 1658|  13.7k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1658:13): [True: 13.7k, False: 0]
  ------------------
 1659|  13.7k|            ptrdiff_t h_off = 0, v_off = 0;
 1660|  13.7k|            if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1660:17): [True: 7.93k, False: 5.81k]
  |  Branch (1660:29): [True: 2.76k, False: 5.16k]
  ------------------
 1661|  8.30k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1661:34): [True: 5.53k, False: 2.76k]
  ------------------
 1662|  5.53k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1663|  5.53k|                             NULL, f->cur.stride[1],
 1664|  5.53k|                             bw4, bh4, t->bx - 1, t->by - 1, 1 + pl,
 1665|  5.53k|                             r[-1][t->bx - 1].mv.mv[0],
 1666|  5.53k|                             &f->refp[r[-1][t->bx - 1].ref.ref[0] - 1],
 1667|  5.53k|                             r[-1][t->bx - 1].ref.ref[0] - 1,
 1668|  5.53k|                             t->frame_thread.pass != 2 ? t->tl_4x4_filter :
  ------------------
  |  Branch (1668:30): [True: 5.53k, False: 0]
  ------------------
 1669|  5.53k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx - 1].filter2d);
 1670|  5.53k|                    if (res) return res;
  ------------------
  |  Branch (1670:25): [True: 0, False: 5.53k]
  ------------------
 1671|  5.53k|                }
 1672|  2.76k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  2.76k|#define PXSTRIDE(x) (x)
  ------------------
 1673|  2.76k|                h_off = 2;
 1674|  2.76k|            }
 1675|  13.7k|            if (bw4 == 1) {
  ------------------
  |  Branch (1675:17): [True: 7.93k, False: 5.81k]
  ------------------
 1676|  7.93k|                const enum Filter2d left_filter_2d =
 1677|  7.93k|                    dav1d_filter_2d[t->l.filter[1][by4]][t->l.filter[0][by4]];
 1678|  23.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1678:34): [True: 15.8k, False: 7.93k]
  ------------------
 1679|  15.8k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + v_off, NULL,
 1680|  15.8k|                             f->cur.stride[1], bw4, bh4, t->bx - 1,
 1681|  15.8k|                             t->by, 1 + pl, r[0][t->bx - 1].mv.mv[0],
 1682|  15.8k|                             &f->refp[r[0][t->bx - 1].ref.ref[0] - 1],
 1683|  15.8k|                             r[0][t->bx - 1].ref.ref[0] - 1,
 1684|  15.8k|                             t->frame_thread.pass != 2 ? left_filter_2d :
  ------------------
  |  Branch (1684:30): [True: 15.8k, False: 0]
  ------------------
 1685|  15.8k|                                 f->frame_thread.b[(t->by * f->b4_stride) + t->bx - 1].filter2d);
 1686|  15.8k|                    if (res) return res;
  ------------------
  |  Branch (1686:25): [True: 0, False: 15.8k]
  ------------------
 1687|  15.8k|                }
 1688|  7.93k|                h_off = 2;
 1689|  7.93k|            }
 1690|  13.7k|            if (bh4 == ss_ver) {
  ------------------
  |  Branch (1690:17): [True: 8.58k, False: 5.16k]
  ------------------
 1691|  8.58k|                const enum Filter2d top_filter_2d =
 1692|  8.58k|                    dav1d_filter_2d[t->a->filter[1][bx4]][t->a->filter[0][bx4]];
 1693|  25.7k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1693:34): [True: 17.1k, False: 8.58k]
  ------------------
 1694|  17.1k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off, NULL,
 1695|  17.1k|                             f->cur.stride[1], bw4, bh4, t->bx, t->by - 1,
 1696|  17.1k|                             1 + pl, r[-1][t->bx].mv.mv[0],
 1697|  17.1k|                             &f->refp[r[-1][t->bx].ref.ref[0] - 1],
 1698|  17.1k|                             r[-1][t->bx].ref.ref[0] - 1,
 1699|  17.1k|                             t->frame_thread.pass != 2 ? top_filter_2d :
  ------------------
  |  Branch (1699:30): [True: 17.1k, False: 0]
  ------------------
 1700|  17.1k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx].filter2d);
 1701|  17.1k|                    if (res) return res;
  ------------------
  |  Branch (1701:25): [True: 0, False: 17.1k]
  ------------------
 1702|  17.1k|                }
 1703|  8.58k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  8.58k|#define PXSTRIDE(x) (x)
  ------------------
 1704|  8.58k|            }
 1705|  41.2k|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1705:30): [True: 27.5k, False: 13.7k]
  ------------------
 1706|  27.5k|                res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off + v_off, NULL, f->cur.stride[1],
 1707|  27.5k|                         bw4, bh4, t->bx, t->by, 1 + pl, b->mv[0],
 1708|  27.5k|                         refp, b->ref[0], filter_2d);
 1709|  27.5k|                if (res) return res;
  ------------------
  |  Branch (1709:21): [True: 0, False: 27.5k]
  ------------------
 1710|  27.5k|            }
 1711|   176k|        } else {
 1712|   176k|            if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (1712:17): [True: 103k, False: 73.8k]
  ------------------
 1713|   103k|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1713:19): [True: 21.5k, False: 81.5k]
  |  Branch (1713:48): [True: 587, False: 20.9k]
  ------------------
 1714|   102k|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1714:19): [True: 12.0k, False: 90.4k]
  |  Branch (1714:48): [True: 10.4k, False: 1.64k]
  ------------------
 1715|  10.9k|            {
 1716|  32.9k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1716:34): [True: 21.9k, False: 10.9k]
  ------------------
 1717|  21.9k|                    res = warp_affine(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff, NULL,
 1718|  21.9k|                                      f->cur.stride[1], b_dim, 1 + pl, refp,
 1719|  21.9k|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1719:39): [True: 20.8k, False: 1.17k]
  ------------------
 1720|  21.9k|                                          &f->frame_hdr->gmv[b->ref[0]]);
 1721|  21.9k|                    if (res) return res;
  ------------------
  |  Branch (1721:25): [True: 0, False: 21.9k]
  ------------------
 1722|  21.9k|                }
 1723|   165k|            } else {
 1724|   497k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1724:34): [True: 331k, False: 165k]
  ------------------
 1725|   331k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1726|   331k|                             NULL, f->cur.stride[1],
 1727|   331k|                             bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1728|   331k|                             t->bx & ~ss_hor, t->by & ~ss_ver,
 1729|   331k|                             1 + pl, b->mv[0], refp, b->ref[0], filter_2d);
 1730|   331k|                    if (res) return res;
  ------------------
  |  Branch (1730:25): [True: 0, False: 331k]
  ------------------
 1731|   331k|                    if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1731:25): [True: 92.6k, False: 239k]
  ------------------
 1732|  92.6k|                        res = obmc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1733|  92.6k|                                   f->cur.stride[1], b_dim, 1 + pl, bx4, by4, w4, h4);
 1734|  92.6k|                        if (res) return res;
  ------------------
  |  Branch (1734:29): [True: 0, False: 92.6k]
  ------------------
 1735|  92.6k|                    }
 1736|   331k|                }
 1737|   165k|            }
 1738|   176k|            if (b->interintra_type) {
  ------------------
  |  Branch (1738:17): [True: 15.6k, False: 161k]
  ------------------
 1739|       |                // FIXME for 8x32 with 4:2:2 subsampling, this probably does
 1740|       |                // the wrong thing since it will select 4x16, not 4x32, as a
 1741|       |                // transform size...
 1742|  15.6k|                const uint8_t *const ii_mask = II_MASK(chr_layout_idx, bs, b);
  ------------------
  |  |   83|  15.6k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  15.6k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 11.7k, False: 3.86k]
  |  |  ------------------
  |  |   85|  15.6k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  15.6k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1743|       |
 1744|  46.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1744:34): [True: 31.2k, False: 15.6k]
  ------------------
 1745|  31.2k|                    pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   51|  31.2k|#define bitfn(x) x##_8bpc
  ------------------
 1746|  31.2k|                    pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   51|  31.2k|#define bitfn(x) x##_8bpc
  ------------------
 1747|  31.2k|                    enum IntraPredMode m =
 1748|  31.2k|                        b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1748:25): [True: 4.64k, False: 26.5k]
  ------------------
 1749|  26.5k|                        SMOOTH_PRED : b->interintra_mode;
 1750|  31.2k|                    int angle = 0;
 1751|  31.2k|                    pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1752|  31.2k|                    const pixel *top_sb_edge = NULL;
 1753|  31.2k|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1753:25): [True: 5.34k, False: 25.8k]
  ------------------
 1754|  5.34k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1755|  5.34k|                        const int sby = t->by >> f->sb_shift;
 1756|  5.34k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1757|  5.34k|                    }
 1758|  31.2k|                    m = bytefn(dav1d_prepare_intra_edges)(t->bx >> ss_hor,
  ------------------
  |  |   87|  31.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  31.2k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1759|  31.2k|                                                          (t->bx >> ss_hor) >
 1760|  31.2k|                                                              (ts->tiling.col_start >> ss_hor),
 1761|  31.2k|                                                          t->by >> ss_ver,
 1762|  31.2k|                                                          (t->by >> ss_ver) >
 1763|  31.2k|                                                              (ts->tiling.row_start >> ss_ver),
 1764|  31.2k|                                                          ts->tiling.col_end >> ss_hor,
 1765|  31.2k|                                                          ts->tiling.row_end >> ss_ver,
 1766|  31.2k|                                                          0, uvdst, f->cur.stride[1],
 1767|  31.2k|                                                          top_sb_edge, m,
 1768|  31.2k|                                                          &angle, cbw4, cbh4, 0, tl_edge
 1769|  31.2k|                                                          HIGHBD_CALL_SUFFIX);
 1770|  31.2k|                    dsp->ipred.intra_pred[m](tmp, cbw4 * 4 * sizeof(pixel),
 1771|  31.2k|                                             tl_edge, cbw4 * 4, cbh4 * 4, 0, 0, 0
 1772|  31.2k|                                             HIGHBD_CALL_SUFFIX);
 1773|  31.2k|                    dsp->mc.blend(uvdst, f->cur.stride[1], tmp,
 1774|  31.2k|                                  cbw4 * 4, cbh4 * 4, ii_mask);
 1775|  31.2k|                }
 1776|  15.6k|            }
 1777|   176k|        }
 1778|       |
 1779|   361k|    skip_inter_chroma_pred: {}
 1780|   361k|        t->tl_4x4_filter = filter_2d;
 1781|   361k|    } else {
 1782|  68.7k|        const enum Filter2d filter_2d = b->filter2d;
 1783|       |        // Maximum super block size is 128x128
 1784|  68.7k|        int16_t (*tmp)[128 * 128] = t->scratch.compinter;
 1785|  68.7k|        int jnt_weight;
 1786|  68.7k|        uint8_t *const seg_mask = t->scratch.seg_mask;
 1787|  68.7k|        const uint8_t *mask;
 1788|       |
 1789|   206k|        for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1789:25): [True: 137k, False: 68.7k]
  ------------------
 1790|   137k|            const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1791|       |
 1792|   137k|            if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (1792:17): [True: 8.80k, False: 128k]
  |  Branch (1792:55): [True: 797, False: 8.00k]
  ------------------
 1793|    797|                res = warp_affine(t, NULL, tmp[i], bw4 * 4, b_dim, 0, refp,
 1794|    797|                                  &f->frame_hdr->gmv[b->ref[i]]);
 1795|    797|                if (res) return res;
  ------------------
  |  Branch (1795:21): [True: 0, False: 797]
  ------------------
 1796|   136k|            } else {
 1797|   136k|                res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by, 0,
 1798|   136k|                         b->mv[i], refp, b->ref[i], filter_2d);
 1799|   136k|                if (res) return res;
  ------------------
  |  Branch (1799:21): [True: 0, False: 136k]
  ------------------
 1800|   136k|            }
 1801|   137k|        }
 1802|  68.7k|        switch (b->comp_type) {
  ------------------
  |  Branch (1802:17): [True: 68.7k, False: 0]
  ------------------
 1803|  43.0k|        case COMP_INTER_AVG:
  ------------------
  |  Branch (1803:9): [True: 43.0k, False: 25.7k]
  ------------------
 1804|  43.0k|            dsp->mc.avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1805|  43.0k|                        bw4 * 4, bh4 * 4 HIGHBD_CALL_SUFFIX);
 1806|  43.0k|            break;
 1807|  8.23k|        case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1807:9): [True: 8.23k, False: 60.5k]
  ------------------
 1808|  8.23k|            jnt_weight = f->jnt_weights[b->ref[0]][b->ref[1]];
 1809|  8.23k|            dsp->mc.w_avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1810|  8.23k|                          bw4 * 4, bh4 * 4, jnt_weight HIGHBD_CALL_SUFFIX);
 1811|  8.23k|            break;
 1812|  13.2k|        case COMP_INTER_SEG:
  ------------------
  |  Branch (1812:9): [True: 13.2k, False: 55.5k]
  ------------------
 1813|  13.2k|            dsp->mc.w_mask[chr_layout_idx](dst, f->cur.stride[0],
 1814|  13.2k|                                           tmp[b->mask_sign], tmp[!b->mask_sign],
 1815|  13.2k|                                           bw4 * 4, bh4 * 4, seg_mask,
 1816|  13.2k|                                           b->mask_sign HIGHBD_CALL_SUFFIX);
 1817|  13.2k|            mask = seg_mask;
 1818|  13.2k|            break;
 1819|  4.32k|        case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1819:9): [True: 4.32k, False: 64.4k]
  ------------------
 1820|  4.32k|            mask = WEDGE_MASK(0, bs, 0, b->wedge_idx);
  ------------------
  |  |   89|  4.32k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  4.32k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1821|  4.32k|            dsp->mc.mask(dst, f->cur.stride[0],
 1822|  4.32k|                         tmp[b->mask_sign], tmp[!b->mask_sign],
 1823|  4.32k|                         bw4 * 4, bh4 * 4, mask HIGHBD_CALL_SUFFIX);
 1824|  4.32k|            if (has_chroma)
  ------------------
  |  Branch (1824:17): [True: 2.28k, False: 2.03k]
  ------------------
 1825|  2.28k|                mask = WEDGE_MASK(chr_layout_idx, bs, b->mask_sign, b->wedge_idx);
  ------------------
  |  |   89|  2.28k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  2.28k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1826|  4.32k|            break;
 1827|  68.7k|        }
 1828|       |
 1829|       |        // chroma
 1830|   117k|        if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1830:13): [True: 39.1k, False: 29.6k]
  |  Branch (1830:42): [True: 78.3k, False: 39.1k]
  ------------------
 1831|   234k|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1831:29): [True: 156k, False: 78.3k]
  ------------------
 1832|   156k|                const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1833|   156k|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (1833:21): [True: 10.4k, False: 146k]
  ------------------
 1834|  10.4k|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (1834:21): [True: 8.57k, False: 1.83k]
  |  Branch (1834:45): [True: 1.30k, False: 7.27k]
  ------------------
 1835|  1.30k|                {
 1836|  1.30k|                    res = warp_affine(t, NULL, tmp[i], bw4 * 4 >> ss_hor,
 1837|  1.30k|                                      b_dim, 1 + pl,
 1838|  1.30k|                                      refp, &f->frame_hdr->gmv[b->ref[i]]);
 1839|  1.30k|                    if (res) return res;
  ------------------
  |  Branch (1839:25): [True: 0, False: 1.30k]
  ------------------
 1840|   155k|                } else {
 1841|   155k|                    res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by,
 1842|   155k|                             1 + pl, b->mv[i], refp, b->ref[i], filter_2d);
 1843|   155k|                    if (res) return res;
  ------------------
  |  Branch (1843:25): [True: 0, False: 155k]
  ------------------
 1844|   155k|                }
 1845|   156k|            }
 1846|  78.3k|            pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1847|  78.3k|            switch (b->comp_type) {
  ------------------
  |  Branch (1847:21): [True: 78.3k, False: 0]
  ------------------
 1848|  45.6k|            case COMP_INTER_AVG:
  ------------------
  |  Branch (1848:13): [True: 45.6k, False: 32.6k]
  ------------------
 1849|  45.6k|                dsp->mc.avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1850|  45.6k|                            bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver
 1851|  45.6k|                            HIGHBD_CALL_SUFFIX);
 1852|  45.6k|                break;
 1853|  13.6k|            case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1853:13): [True: 13.6k, False: 64.6k]
  ------------------
 1854|  13.6k|                dsp->mc.w_avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1855|  13.6k|                              bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, jnt_weight
 1856|  13.6k|                              HIGHBD_CALL_SUFFIX);
 1857|  13.6k|                break;
 1858|  4.57k|            case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1858:13): [True: 4.57k, False: 73.7k]
  ------------------
 1859|  18.9k|            case COMP_INTER_SEG:
  ------------------
  |  Branch (1859:13): [True: 14.4k, False: 63.8k]
  ------------------
 1860|  18.9k|                dsp->mc.mask(uvdst, f->cur.stride[1],
 1861|  18.9k|                             tmp[b->mask_sign], tmp[!b->mask_sign],
 1862|  18.9k|                             bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, mask
 1863|  18.9k|                             HIGHBD_CALL_SUFFIX);
 1864|  18.9k|                break;
 1865|  78.3k|            }
 1866|  78.3k|        }
 1867|  68.7k|    }
 1868|       |
 1869|   593k|    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|   593k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 593k]
  |  |  ------------------
  |  |   35|   593k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   593k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1870|      0|        hex_dump(dst, f->cur.stride[0], b_dim[0] * 4, b_dim[1] * 4, "y-pred");
 1871|      0|        if (has_chroma) {
  ------------------
  |  Branch (1871:13): [True: 0, False: 0]
  ------------------
 1872|      0|            hex_dump(&((pixel *) f->cur.data[1])[uvdstoff], f->cur.stride[1],
 1873|      0|                     cbw4 * 4, cbh4 * 4, "u-pred");
 1874|      0|            hex_dump(&((pixel *) f->cur.data[2])[uvdstoff], f->cur.stride[1],
 1875|      0|                     cbw4 * 4, cbh4 * 4, "v-pred");
 1876|      0|        }
 1877|      0|    }
 1878|       |
 1879|   593k|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1880|       |
 1881|   593k|    if (b->skip) {
  ------------------
  |  Branch (1881:9): [True: 369k, False: 224k]
  ------------------
 1882|       |        // reset coef contexts
 1883|   369k|        BlockContext *const a = t->a;
 1884|   369k|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
 1885|   369k|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
 1886|   369k|        if (has_chroma) {
  ------------------
  |  Branch (1886:13): [True: 208k, False: 161k]
  ------------------
 1887|   208k|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
 1888|   208k|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
 1889|   208k|            memset_cw(&a->ccoef[0][cbx4], 0x40);
 1890|   208k|            memset_cw(&a->ccoef[1][cbx4], 0x40);
 1891|   208k|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
 1892|   208k|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
 1893|   208k|        }
 1894|   369k|        return 0;
 1895|   369k|    }
 1896|       |
 1897|   224k|    const TxfmInfo *const uvtx = &dav1d_txfm_dimensions[b->uvtx];
 1898|   224k|    const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
 1899|   224k|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1900|       |
 1901|   452k|    for (int init_y = 0; init_y < bh4; init_y += 16) {
  ------------------
  |  Branch (1901:26): [True: 227k, False: 224k]
  ------------------
 1902|   462k|        for (int init_x = 0; init_x < bw4; init_x += 16) {
  ------------------
  |  Branch (1902:30): [True: 234k, False: 227k]
  ------------------
 1903|       |            // coefficient coding & inverse transforms
 1904|   234k|            int y_off = !!init_y, y;
 1905|   234k|            dst += PXSTRIDE(f->cur.stride[0]) * 4 * init_y;
  ------------------
  |  |   53|   234k|#define PXSTRIDE(x) (x)
  ------------------
 1906|   499k|            for (y = init_y, t->by += init_y; y < imin(h4, init_y + 16);
  ------------------
  |  Branch (1906:47): [True: 265k, False: 234k]
  ------------------
 1907|   265k|                 y += ytx->h, y_off++)
 1908|   265k|            {
 1909|   265k|                int x, x_off = !!init_x;
 1910|   690k|                for (x = init_x, t->bx += init_x; x < imin(w4, init_x + 16);
  ------------------
  |  Branch (1910:51): [True: 424k, False: 265k]
  ------------------
 1911|   424k|                     x += ytx->w, x_off++)
 1912|   424k|                {
 1913|   424k|                    read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
 1914|   424k|                                   x_off, y_off, &dst[x * 4]);
 1915|   424k|                    t->bx += ytx->w;
 1916|   424k|                }
 1917|   265k|                dst += PXSTRIDE(f->cur.stride[0]) * 4 * ytx->h;
  ------------------
  |  |   53|   265k|#define PXSTRIDE(x) (x)
  ------------------
 1918|   265k|                t->bx -= x;
 1919|   265k|                t->by += ytx->h;
 1920|   265k|            }
 1921|   234k|            dst -= PXSTRIDE(f->cur.stride[0]) * 4 * y;
  ------------------
  |  |   53|   234k|#define PXSTRIDE(x) (x)
  ------------------
 1922|   234k|            t->by -= y;
 1923|       |
 1924|       |            // chroma coefs and inverse transform
 1925|   480k|            if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1925:17): [True: 160k, False: 74.5k]
  |  Branch (1925:46): [True: 320k, False: 160k]
  ------------------
 1926|   320k|                pixel *uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff +
 1927|   320k|                    (PXSTRIDE(f->cur.stride[1]) * init_y * 4 >> ss_ver);
  ------------------
  |  |   53|   320k|#define PXSTRIDE(x) (x)
  ------------------
 1928|   320k|                for (y = init_y >> ss_ver, t->by += init_y;
 1929|   670k|                     y < imin(ch4, (init_y + 16) >> ss_ver); y += uvtx->h)
  ------------------
  |  Branch (1929:22): [True: 350k, False: 320k]
  ------------------
 1930|   350k|                {
 1931|   350k|                    int x;
 1932|   350k|                    for (x = init_x >> ss_hor, t->bx += init_x;
 1933|   789k|                         x < imin(cw4, (init_x + 16) >> ss_hor); x += uvtx->w)
  ------------------
  |  Branch (1933:26): [True: 439k, False: 350k]
  ------------------
 1934|   439k|                    {
 1935|   439k|                        coef *cf;
 1936|   439k|                        int eob;
 1937|   439k|                        enum TxfmType txtp;
 1938|   439k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1938:29): [True: 0, False: 439k]
  ------------------
 1939|      0|                            const int p = t->frame_thread.pass & 1;
 1940|      0|                            const int cbi = *ts->frame_thread[p].cbi++;
 1941|      0|                            cf = ts->frame_thread[p].cf;
 1942|      0|                            ts->frame_thread[p].cf += uvtx->w * uvtx->h * 16;
 1943|      0|                            eob  = cbi >> 5;
 1944|      0|                            txtp = cbi & 0x1f;
 1945|   439k|                        } else {
 1946|   439k|                            uint8_t cf_ctx;
 1947|   439k|                            cf = bitfn(t->cf);
  ------------------
  |  |   51|   439k|#define bitfn(x) x##_8bpc
  ------------------
 1948|   439k|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
 1949|   439k|                                                        bx4 + (x << ss_hor)];
 1950|   439k|                            eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1951|   439k|                                               &t->l.ccoef[pl][cby4 + y],
 1952|   439k|                                               b->uvtx, bs, b, 0, 1 + pl,
 1953|   439k|                                               cf, &txtp, &cf_ctx);
 1954|   439k|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   439k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 439k]
  |  |  ------------------
  |  |   35|   439k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   439k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1955|      0|                                printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1956|      0|                                       "txtp=%d,eob=%d]: r=%d\n",
 1957|      0|                                       pl, b->uvtx, txtp, eob, ts->msac.rng);
 1958|   439k|                            int ctw = imin(uvtx->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1959|   439k|                            int cth = imin(uvtx->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1960|   439k|                            dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1961|   439k|                            dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1962|   439k|                        }
 1963|   439k|                        if (eob >= 0) {
  ------------------
  |  Branch (1963:29): [True: 173k, False: 265k]
  ------------------
 1964|   173k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   173k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 173k]
  |  |  ------------------
  |  |   35|   173k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   173k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1965|      0|                                coef_dump(cf, uvtx->h * 4, uvtx->w * 4, 3, "dq");
 1966|   173k|                            dsp->itx.itxfm_add[b->uvtx]
 1967|   173k|                                              [txtp](&uvdst[4 * x],
 1968|   173k|                                                     f->cur.stride[1],
 1969|   173k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
 1970|   173k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   173k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 173k]
  |  |  ------------------
  |  |   35|   173k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   173k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1971|      0|                                hex_dump(&uvdst[4 * x], f->cur.stride[1],
 1972|      0|                                         uvtx->w * 4, uvtx->h * 4, "recon");
 1973|   173k|                        }
 1974|   439k|                        t->bx += uvtx->w << ss_hor;
 1975|   439k|                    }
 1976|   350k|                    uvdst += PXSTRIDE(f->cur.stride[1]) * 4 * uvtx->h;
  ------------------
  |  |   53|   350k|#define PXSTRIDE(x) (x)
  ------------------
 1977|   350k|                    t->bx -= x << ss_hor;
 1978|   350k|                    t->by += uvtx->h << ss_ver;
 1979|   350k|                }
 1980|   320k|                t->by -= y << ss_ver;
 1981|   320k|            }
 1982|   234k|        }
 1983|   227k|    }
 1984|   224k|    return 0;
 1985|   593k|}
dav1d_filter_sbrow_deblock_cols_8bpc:
 1987|  74.0k|void bytefn(dav1d_filter_sbrow_deblock_cols)(Dav1dFrameContext *const f, const int sby) {
 1988|  74.0k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK) ||
  ------------------
  |  Branch (1988:9): [True: 0, False: 74.0k]
  ------------------
 1989|  74.0k|        (!f->frame_hdr->loopfilter.level_y[0] && !f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (1989:10): [True: 50.2k, False: 23.7k]
  |  Branch (1989:50): [True: 39.7k, False: 10.4k]
  ------------------
 1990|  39.7k|    {
 1991|  39.7k|        return;
 1992|  39.7k|    }
 1993|  34.2k|    const int y = sby * f->sb_step * 4;
 1994|  34.2k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1995|  34.2k|    pixel *const p[3] = {
 1996|  34.2k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  34.2k|#define PXSTRIDE(x) (x)
  ------------------
 1997|  34.2k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  34.2k|#define PXSTRIDE(x) (x)
  ------------------
 1998|  34.2k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  34.2k|#define PXSTRIDE(x) (x)
  ------------------
 1999|  34.2k|    };
 2000|  34.2k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2001|  34.2k|    bytefn(dav1d_loopfilter_sbrow_cols)(f, p, mask, sby,
  ------------------
  |  |   87|  34.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  34.2k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2002|  34.2k|                                        f->lf.start_of_tile_row[sby]);
 2003|  34.2k|}
dav1d_filter_sbrow_deblock_rows_8bpc:
 2005|  74.0k|void bytefn(dav1d_filter_sbrow_deblock_rows)(Dav1dFrameContext *const f, const int sby) {
 2006|  74.0k|    const int y = sby * f->sb_step * 4;
 2007|  74.0k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2008|  74.0k|    pixel *const p[3] = {
 2009|  74.0k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  74.0k|#define PXSTRIDE(x) (x)
  ------------------
 2010|  74.0k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  74.0k|#define PXSTRIDE(x) (x)
  ------------------
 2011|  74.0k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  74.0k|#define PXSTRIDE(x) (x)
  ------------------
 2012|  74.0k|    };
 2013|  74.0k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2014|  74.0k|    if (f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK &&
  ------------------
  |  Branch (2014:9): [True: 74.0k, False: 0]
  ------------------
 2015|  74.0k|        (f->frame_hdr->loopfilter.level_y[0] || f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (2015:10): [True: 23.7k, False: 50.2k]
  |  Branch (2015:49): [True: 10.4k, False: 39.7k]
  ------------------
 2016|  34.2k|    {
 2017|  34.2k|        bytefn(dav1d_loopfilter_sbrow_rows)(f, p, mask, sby);
  ------------------
  |  |   87|  34.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  34.2k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2018|  34.2k|    }
 2019|  74.0k|    if (f->seq_hdr->cdef || f->lf.restore_planes) {
  ------------------
  |  Branch (2019:9): [True: 34.6k, False: 39.4k]
  |  Branch (2019:29): [True: 11.7k, False: 27.6k]
  ------------------
 2020|       |        // Store loop filtered pixels required by CDEF / LR
 2021|  46.3k|        bytefn(dav1d_copy_lpf)(f, p, sby);
  ------------------
  |  |   87|  46.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  46.3k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2022|  46.3k|    }
 2023|  74.0k|}
dav1d_filter_sbrow_cdef_8bpc:
 2025|  34.6k|void bytefn(dav1d_filter_sbrow_cdef)(Dav1dTaskContext *const tc, const int sby) {
 2026|  34.6k|    const Dav1dFrameContext *const f = tc->f;
 2027|  34.6k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_CDEF)) return;
  ------------------
  |  Branch (2027:9): [True: 0, False: 34.6k]
  ------------------
 2028|  34.6k|    const int sbsz = f->sb_step;
 2029|  34.6k|    const int y = sby * sbsz * 4;
 2030|  34.6k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2031|  34.6k|    pixel *const p[3] = {
 2032|  34.6k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  34.6k|#define PXSTRIDE(x) (x)
  ------------------
 2033|  34.6k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  34.6k|#define PXSTRIDE(x) (x)
  ------------------
 2034|  34.6k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  34.6k|#define PXSTRIDE(x) (x)
  ------------------
 2035|  34.6k|    };
 2036|  34.6k|    Av1Filter *prev_mask = f->lf.mask + ((sby - 1) >> !f->seq_hdr->sb128) * f->sb128w;
 2037|  34.6k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2038|  34.6k|    const int start = sby * sbsz;
 2039|  34.6k|    if (sby) {
  ------------------
  |  Branch (2039:9): [True: 31.6k, False: 2.97k]
  ------------------
 2040|  31.6k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2041|  31.6k|        pixel *p_up[3] = {
 2042|  31.6k|            p[0] - 8 * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  31.6k|#define PXSTRIDE(x) (x)
  ------------------
 2043|  31.6k|            p[1] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  31.6k|#define PXSTRIDE(x) (x)
  ------------------
 2044|  31.6k|            p[2] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  31.6k|#define PXSTRIDE(x) (x)
  ------------------
 2045|  31.6k|        };
 2046|  31.6k|        bytefn(dav1d_cdef_brow)(tc, p_up, prev_mask, start - 2, start, 1, sby);
  ------------------
  |  |   87|  31.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  31.6k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2047|  31.6k|    }
 2048|  34.6k|    const int n_blks = sbsz - 2 * (sby + 1 < f->sbh);
 2049|  34.6k|    const int end = imin(start + n_blks, f->bh);
 2050|  34.6k|    bytefn(dav1d_cdef_brow)(tc, p, mask, start, end, 0, sby);
  ------------------
  |  |   87|  34.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  34.6k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2051|  34.6k|}
dav1d_filter_sbrow_resize_8bpc:
 2053|  4.00k|void bytefn(dav1d_filter_sbrow_resize)(Dav1dFrameContext *const f, const int sby) {
 2054|  4.00k|    const int sbsz = f->sb_step;
 2055|  4.00k|    const int y = sby * sbsz * 4;
 2056|  4.00k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2057|  4.00k|    const pixel *const p[3] = {
 2058|  4.00k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2059|  4.00k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2060|  4.00k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2061|  4.00k|    };
 2062|  4.00k|    pixel *const sr_p[3] = {
 2063|  4.00k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2064|  4.00k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2065|  4.00k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  4.00k|#define PXSTRIDE(x) (x)
  ------------------
 2066|  4.00k|    };
 2067|  4.00k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 2068|  13.4k|    for (int pl = 0; pl < 1 + 2 * has_chroma; pl++) {
  ------------------
  |  Branch (2068:22): [True: 9.39k, False: 4.00k]
  ------------------
 2069|  9.39k|        const int ss_ver = pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2069:28): [True: 5.39k, False: 4.00k]
  |  Branch (2069:34): [True: 1.46k, False: 3.92k]
  ------------------
 2070|  9.39k|        const int h_start = 8 * !!sby >> ss_ver;
 2071|  9.39k|        const ptrdiff_t dst_stride = f->sr_cur.p.stride[!!pl];
 2072|  9.39k|        pixel *dst = sr_p[pl] - h_start * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  9.39k|#define PXSTRIDE(x) (x)
  ------------------
 2073|  9.39k|        const ptrdiff_t src_stride = f->cur.stride[!!pl];
 2074|  9.39k|        const pixel *src = p[pl] - h_start * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  9.39k|#define PXSTRIDE(x) (x)
  ------------------
 2075|  9.39k|        const int h_end = 4 * (sbsz - 2 * (sby + 1 < f->sbh)) >> ss_ver;
 2076|  9.39k|        const int ss_hor = pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2076:28): [True: 5.39k, False: 4.00k]
  |  Branch (2076:34): [True: 4.24k, False: 1.14k]
  ------------------
 2077|  9.39k|        const int dst_w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2078|  9.39k|        const int src_w = (4 * f->bw + ss_hor) >> ss_hor;
 2079|  9.39k|        const int img_h = (f->cur.p.h - sbsz * 4 * sby + ss_ver) >> ss_ver;
 2080|       |
 2081|  9.39k|        f->dsp->mc.resize(dst, dst_stride, src, src_stride, dst_w,
 2082|  9.39k|                          imin(img_h, h_end) + h_start, src_w,
 2083|  9.39k|                          f->resize_step[!!pl], f->resize_start[!!pl]
 2084|  9.39k|                          HIGHBD_CALL_SUFFIX);
 2085|  9.39k|    }
 2086|  4.00k|}
dav1d_filter_sbrow_lr_8bpc:
 2088|  21.3k|void bytefn(dav1d_filter_sbrow_lr)(Dav1dFrameContext *const f, const int sby) {
 2089|  21.3k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_RESTORATION)) return;
  ------------------
  |  Branch (2089:9): [True: 0, False: 21.3k]
  ------------------
 2090|  21.3k|    const int y = sby * f->sb_step * 4;
 2091|  21.3k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2092|  21.3k|    pixel *const sr_p[3] = {
 2093|  21.3k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
  ------------------
  |  |   53|  21.3k|#define PXSTRIDE(x) (x)
  ------------------
 2094|  21.3k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  21.3k|#define PXSTRIDE(x) (x)
  ------------------
 2095|  21.3k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  21.3k|#define PXSTRIDE(x) (x)
  ------------------
 2096|  21.3k|    };
 2097|  21.3k|    bytefn(dav1d_lr_sbrow)(f, sr_p, sby);
  ------------------
  |  |   87|  21.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  21.3k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2098|  21.3k|}
dav1d_filter_sbrow_8bpc:
 2100|  74.0k|void bytefn(dav1d_filter_sbrow)(Dav1dFrameContext *const f, const int sby) {
 2101|  74.0k|    bytefn(dav1d_filter_sbrow_deblock_cols)(f, sby);
  ------------------
  |  |   87|  74.0k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  74.0k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2102|  74.0k|    bytefn(dav1d_filter_sbrow_deblock_rows)(f, sby);
  ------------------
  |  |   87|  74.0k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  74.0k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2103|  74.0k|    if (f->seq_hdr->cdef)
  ------------------
  |  Branch (2103:9): [True: 34.6k, False: 39.4k]
  ------------------
 2104|  34.6k|        bytefn(dav1d_filter_sbrow_cdef)(f->c->tc, sby);
  ------------------
  |  |   87|  34.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  34.6k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2105|  74.0k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1])
  ------------------
  |  Branch (2105:9): [True: 4.00k, False: 70.0k]
  ------------------
 2106|  4.00k|        bytefn(dav1d_filter_sbrow_resize)(f, sby);
  ------------------
  |  |   87|  4.00k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  4.00k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2107|  74.0k|    if (f->lf.restore_planes)
  ------------------
  |  Branch (2107:9): [True: 21.3k, False: 52.7k]
  ------------------
 2108|  21.3k|        bytefn(dav1d_filter_sbrow_lr)(f, sby);
  ------------------
  |  |   87|  21.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  21.3k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2109|  74.0k|}
dav1d_backup_ipred_edge_8bpc:
 2111|  81.1k|void bytefn(dav1d_backup_ipred_edge)(Dav1dTaskContext *const t) {
 2112|  81.1k|    const Dav1dFrameContext *const f = t->f;
 2113|  81.1k|    Dav1dTileState *const ts = t->ts;
 2114|  81.1k|    const int sby = t->by >> f->sb_shift;
 2115|  81.1k|    const int sby_off = f->sb128w * 128 * sby;
 2116|  81.1k|    const int x_off = ts->tiling.col_start;
 2117|       |
 2118|  81.1k|    const pixel *const y =
 2119|  81.1k|        ((const pixel *) f->cur.data[0]) + x_off * 4 +
 2120|  81.1k|                    ((t->by + f->sb_step) * 4 - 1) * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|  81.1k|#define PXSTRIDE(x) (x)
  ------------------
 2121|  81.1k|    pixel_copy(&f->ipred_edge[0][sby_off + x_off * 4], y,
  ------------------
  |  |   47|  81.1k|#define pixel_copy memcpy
  ------------------
 2122|  81.1k|               4 * (ts->tiling.col_end - x_off));
 2123|       |
 2124|  81.1k|    if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (2124:9): [True: 34.3k, False: 46.7k]
  ------------------
 2125|  34.3k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2126|  34.3k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 2127|       |
 2128|  34.3k|        const ptrdiff_t uv_off = (x_off * 4 >> ss_hor) +
 2129|  34.3k|            (((t->by + f->sb_step) * 4 >> ss_ver) - 1) * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  34.3k|#define PXSTRIDE(x) (x)
  ------------------
 2130|   103k|        for (int pl = 1; pl <= 2; pl++)
  ------------------
  |  Branch (2130:26): [True: 68.7k, False: 34.3k]
  ------------------
 2131|  68.7k|            pixel_copy(&f->ipred_edge[pl][sby_off + (x_off * 4 >> ss_hor)],
  ------------------
  |  |   47|  68.7k|#define pixel_copy memcpy
  ------------------
 2132|  68.7k|                       &((const pixel *) f->cur.data[pl])[uv_off],
 2133|  68.7k|                       4 * (ts->tiling.col_end - x_off) >> ss_hor);
 2134|  34.3k|    }
 2135|  81.1k|}
dav1d_copy_pal_block_y_8bpc:
 2141|  21.0k|{
 2142|  21.0k|    const Dav1dFrameContext *const f = t->f;
 2143|  21.0k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2143:24): [True: 0, False: 21.0k]
  ------------------
 2144|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2145|      0|                            ((t->bx >> 1) + (t->by & 1))][0] :
 2146|  21.0k|        bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|  21.0k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  21.0k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2147|  93.5k|    for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2147:21): [True: 72.5k, False: 21.0k]
  ------------------
 2148|  72.5k|        memcpy(bytefn(t->al_pal)[0][bx4 + x][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|  72.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  72.5k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2149|  86.3k|    for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2149:21): [True: 65.3k, False: 21.0k]
  ------------------
 2150|  65.3k|        memcpy(bytefn(t->al_pal)[1][by4 + y][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|  65.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  65.3k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2151|  21.0k|}
dav1d_copy_pal_block_uv_8bpc:
 2157|  4.92k|{
 2158|  4.92k|    const Dav1dFrameContext *const f = t->f;
 2159|  4.92k|    const pixel (*const pal)[8] = t->frame_thread.pass ?
  ------------------
  |  Branch (2159:35): [True: 0, False: 4.92k]
  ------------------
 2160|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2161|      0|                            ((t->bx >> 1) + (t->by & 1))] :
 2162|  4.92k|        bytefn(t->scratch.pal);
  ------------------
  |  |   87|  4.92k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  4.92k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2163|       |    // see aomedia bug 2183 for why we use luma coordinates here
 2164|  14.7k|    for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (2164:22): [True: 9.84k, False: 4.92k]
  ------------------
 2165|  42.9k|        for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2165:25): [True: 33.1k, False: 9.84k]
  ------------------
 2166|  33.1k|            memcpy(bytefn(t->al_pal)[0][bx4 + x][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|  33.1k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  33.1k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2167|  42.8k|        for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2167:25): [True: 32.9k, False: 9.84k]
  ------------------
 2168|  32.9k|            memcpy(bytefn(t->al_pal)[1][by4 + y][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|  32.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  32.9k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2169|  9.84k|    }
 2170|  4.92k|}
dav1d_read_pal_plane_8bpc:
 2175|  25.9k|{
 2176|  25.9k|    Dav1dTileState *const ts = t->ts;
 2177|  25.9k|    const Dav1dFrameContext *const f = t->f;
 2178|  25.9k|    const int pal_sz = b->pal_sz[pl] = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  25.9k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 2179|  25.9k|                                           ts->cdf.m.pal_sz[pl][sz_ctx], 6) + 2;
 2180|  25.9k|    pixel cache[16], used_cache[8];
 2181|  25.9k|    int l_cache = pl ? t->pal_sz_uv[1][by4] : t->l.pal_sz[by4];
  ------------------
  |  Branch (2181:19): [True: 4.92k, False: 21.0k]
  ------------------
 2182|  25.9k|    int n_cache = 0;
 2183|       |    // don't reuse above palette outside SB64 boundaries
 2184|  25.9k|    int a_cache = by4 & 15 ? pl ? t->pal_sz_uv[0][bx4] : t->a->pal_sz[bx4] : 0;
  ------------------
  |  Branch (2184:19): [True: 22.9k, False: 2.97k]
  |  Branch (2184:30): [True: 4.43k, False: 18.5k]
  ------------------
 2185|  25.9k|    const pixel *l = bytefn(t->al_pal)[1][by4][pl];
  ------------------
  |  |   87|  25.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  25.9k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2186|  25.9k|    const pixel *a = bytefn(t->al_pal)[0][bx4][pl];
  ------------------
  |  |   87|  25.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  25.9k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2187|       |
 2188|       |    // fill/sort cache
 2189|  54.4k|    while (l_cache && a_cache) {
  ------------------
  |  Branch (2189:12): [True: 37.6k, False: 16.7k]
  |  Branch (2189:23): [True: 28.5k, False: 9.15k]
  ------------------
 2190|  28.5k|        if (*l < *a) {
  ------------------
  |  Branch (2190:13): [True: 9.49k, False: 19.0k]
  ------------------
 2191|  9.49k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2191:17): [True: 1.05k, False: 8.44k]
  |  Branch (2191:29): [True: 8.05k, False: 393]
  ------------------
 2192|  9.10k|                cache[n_cache++] = *l;
 2193|  9.49k|            l++;
 2194|  9.49k|            l_cache--;
 2195|  19.0k|        } else {
 2196|  19.0k|            if (*a == *l) {
  ------------------
  |  Branch (2196:17): [True: 8.73k, False: 10.2k]
  ------------------
 2197|  8.73k|                l++;
 2198|  8.73k|                l_cache--;
 2199|  8.73k|            }
 2200|  19.0k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2200:17): [True: 3.59k, False: 15.4k]
  |  Branch (2200:29): [True: 14.0k, False: 1.37k]
  ------------------
 2201|  17.6k|                cache[n_cache++] = *a;
 2202|  19.0k|            a++;
 2203|  19.0k|            a_cache--;
 2204|  19.0k|        }
 2205|  28.5k|    }
 2206|  25.9k|    if (l_cache) {
  ------------------
  |  Branch (2206:9): [True: 9.15k, False: 16.7k]
  ------------------
 2207|  36.2k|        do {
 2208|  36.2k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2208:17): [True: 6.99k, False: 29.2k]
  |  Branch (2208:29): [True: 24.1k, False: 5.11k]
  ------------------
 2209|  31.1k|                cache[n_cache++] = *l;
 2210|  36.2k|            l++;
 2211|  36.2k|        } while (--l_cache > 0);
  ------------------
  |  Branch (2211:18): [True: 27.1k, False: 9.15k]
  ------------------
 2212|  16.7k|    } else if (a_cache) {
  ------------------
  |  Branch (2212:16): [True: 6.43k, False: 10.3k]
  ------------------
 2213|  27.9k|        do {
 2214|  27.9k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2214:17): [True: 4.78k, False: 23.1k]
  |  Branch (2214:29): [True: 17.2k, False: 5.90k]
  ------------------
 2215|  22.0k|                cache[n_cache++] = *a;
 2216|  27.9k|            a++;
 2217|  27.9k|        } while (--a_cache > 0);
  ------------------
  |  Branch (2217:18): [True: 21.5k, False: 6.43k]
  ------------------
 2218|  6.43k|    }
 2219|       |
 2220|       |    // find reused cache entries
 2221|  25.9k|    int i = 0;
 2222|  96.4k|    for (int n = 0; n < n_cache && i < pal_sz; n++)
  ------------------
  |  Branch (2222:21): [True: 73.8k, False: 22.6k]
  |  Branch (2222:36): [True: 70.4k, False: 3.34k]
  ------------------
 2223|  70.4k|        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|  70.4k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2223:13): [True: 36.2k, False: 34.2k]
  ------------------
 2224|  36.2k|            used_cache[i++] = cache[n];
 2225|  25.9k|    const int n_used_cache = i;
 2226|       |
 2227|       |    // parse new entries
 2228|  25.9k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2228:24): [True: 0, False: 25.9k]
  ------------------
 2229|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2230|      0|                            ((t->bx >> 1) + (t->by & 1))][pl] :
 2231|  25.9k|        bytefn(t->scratch.pal)[pl];
  ------------------
  |  |   87|  25.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  25.9k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2232|  25.9k|    if (i < pal_sz) {
  ------------------
  |  Branch (2232:9): [True: 21.2k, False: 4.67k]
  ------------------
 2233|  21.2k|        const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2233:25): [True: 21.2k, Folded]
  ------------------
 2234|  21.2k|        int prev = pal[i++] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2235|       |
 2236|  21.2k|        if (i < pal_sz) {
  ------------------
  |  Branch (2236:13): [True: 18.0k, False: 3.19k]
  ------------------
 2237|  18.0k|            int bits = bpc - 3 + dav1d_msac_decode_bools(&ts->msac, 2);
 2238|  18.0k|            const int max = (1 << bpc) - 1;
 2239|       |
 2240|  41.5k|            do {
 2241|  41.5k|                const int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2242|  41.5k|                prev = pal[i++] = imin(prev + delta + !pl, max);
 2243|  41.5k|                if (prev + !pl >= max) {
  ------------------
  |  Branch (2243:21): [True: 8.64k, False: 32.9k]
  ------------------
 2244|  22.5k|                    for (; i < pal_sz; i++)
  ------------------
  |  Branch (2244:28): [True: 13.8k, False: 8.64k]
  ------------------
 2245|  13.8k|                        pal[i] = max;
 2246|  8.64k|                    break;
 2247|  8.64k|                }
 2248|  32.9k|                bits = imin(bits, 1 + ulog2(max - prev - !pl));
 2249|  32.9k|            } while (i < pal_sz);
  ------------------
  |  Branch (2249:22): [True: 23.4k, False: 9.44k]
  ------------------
 2250|  18.0k|        }
 2251|       |
 2252|       |        // merge cache+new entries
 2253|  21.2k|        int n = 0, m = n_used_cache;
 2254|   120k|        for (i = 0; i < pal_sz; i++) {
  ------------------
  |  Branch (2254:21): [True: 99.0k, False: 21.2k]
  ------------------
 2255|  99.0k|            if (n < n_used_cache && (m >= pal_sz || used_cache[n] <= pal[m])) {
  ------------------
  |  Branch (2255:17): [True: 35.8k, False: 63.2k]
  |  Branch (2255:38): [True: 7.34k, False: 28.4k]
  |  Branch (2255:53): [True: 15.0k, False: 13.4k]
  ------------------
 2256|  22.3k|                pal[i] = used_cache[n++];
 2257|  76.7k|            } else {
 2258|  76.7k|                assert(m < pal_sz);
  ------------------
  |  Branch (2258:17): [True: 76.7k, False: 0]
  ------------------
 2259|  76.7k|                pal[i] = pal[m++];
 2260|  76.7k|            }
 2261|  99.0k|        }
 2262|  21.2k|    } else {
 2263|  4.67k|        memcpy(pal, used_cache, n_used_cache * sizeof(*used_cache));
 2264|  4.67k|    }
 2265|       |
 2266|  25.9k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  25.9k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 25.9k]
  |  |  ------------------
  |  |   35|  25.9k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  25.9k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2267|      0|        printf("Post-pal[pl=%d,sz=%d,cache_size=%d,used_cache=%d]: r=%d, cache=",
 2268|      0|               pl, pal_sz, n_cache, n_used_cache, ts->msac.rng);
 2269|      0|        for (int n = 0; n < n_cache; n++)
  ------------------
  |  Branch (2269:25): [True: 0, False: 0]
  ------------------
 2270|      0|            printf("%c%02x", n ? ' ' : '[', cache[n]);
  ------------------
  |  Branch (2270:30): [True: 0, False: 0]
  ------------------
 2271|      0|        printf("%s, pal=", n_cache ? "]" : "[]");
  ------------------
  |  Branch (2271:28): [True: 0, False: 0]
  ------------------
 2272|      0|        for (int n = 0; n < pal_sz; n++)
  ------------------
  |  Branch (2272:25): [True: 0, False: 0]
  ------------------
 2273|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2273:30): [True: 0, False: 0]
  ------------------
 2274|      0|        printf("]\n");
 2275|      0|    }
 2276|  25.9k|}
dav1d_read_pal_uv_8bpc:
 2280|  4.92k|{
 2281|  4.92k|    bytefn(dav1d_read_pal_plane)(t, b, 1, sz_ctx, bx4, by4);
  ------------------
  |  |   87|  4.92k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  4.92k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2282|       |
 2283|       |    // V pal coding
 2284|  4.92k|    Dav1dTileState *const ts = t->ts;
 2285|  4.92k|    const Dav1dFrameContext *const f = t->f;
 2286|  4.92k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2286:24): [True: 0, False: 4.92k]
  ------------------
 2287|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2288|      0|                            ((t->bx >> 1) + (t->by & 1))][2] :
 2289|  4.92k|        bytefn(t->scratch.pal)[2];
  ------------------
  |  |   87|  4.92k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  4.92k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2290|  4.92k|    const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2290:21): [True: 4.92k, Folded]
  ------------------
 2291|  4.92k|    if (dav1d_msac_decode_bool_equi(&ts->msac)) {
  ------------------
  |  |   53|  4.92k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2291:9): [True: 3.04k, False: 1.87k]
  ------------------
 2292|  3.04k|        const int bits = bpc - 4 + dav1d_msac_decode_bools(&ts->msac, 2);
 2293|  3.04k|        int prev = pal[0] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2294|  3.04k|        const int max = (1 << bpc) - 1;
 2295|  11.8k|        for (int i = 1; i < b->pal_sz[1]; i++) {
  ------------------
  |  Branch (2295:25): [True: 8.78k, False: 3.04k]
  ------------------
 2296|  8.78k|            int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2297|  8.78k|            if (delta && dav1d_msac_decode_bool_equi(&ts->msac)) delta = -delta;
  ------------------
  |  |   53|  8.33k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2297:17): [True: 8.33k, False: 448]
  |  Branch (2297:26): [True: 3.09k, False: 5.23k]
  ------------------
 2298|  8.78k|            prev = pal[i] = (prev + delta) & max;
 2299|  8.78k|        }
 2300|  3.04k|    } else {
 2301|  8.34k|        for (int i = 0; i < b->pal_sz[1]; i++)
  ------------------
  |  Branch (2301:25): [True: 6.46k, False: 1.87k]
  ------------------
 2302|  6.46k|            pal[i] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2303|  1.87k|    }
 2304|  4.92k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  4.92k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 4.92k]
  |  |  ------------------
  |  |   35|  4.92k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  4.92k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2305|      0|        printf("Post-pal[pl=2]: r=%d ", ts->msac.rng);
 2306|      0|        for (int n = 0; n < b->pal_sz[1]; n++)
  ------------------
  |  Branch (2306:25): [True: 0, False: 0]
  ------------------
 2307|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2307:30): [True: 0, False: 0]
  ------------------
 2308|      0|        printf("]\n");
 2309|      0|    }
 2310|  4.92k|}
recon_tmpl.c:read_coef_tree:
  736|  1.06M|{
  737|  1.06M|    const Dav1dFrameContext *const f = t->f;
  738|  1.06M|    Dav1dTileState *const ts = t->ts;
  739|  1.06M|    const Dav1dDSPContext *const dsp = f->dsp;
  740|  1.06M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[ytx];
  741|  1.06M|    const int txw = t_dim->w, txh = t_dim->h;
  742|       |
  743|       |    /* y_off can be larger than 3 since lossless blocks use TX_4X4 but can't
  744|       |     * be splitted. Aviods an undefined left shift. */
  745|  1.06M|    if (depth < 2 && tx_split[depth] &&
  ------------------
  |  Branch (745:9): [True: 979k, False: 84.6k]
  |  Branch (745:22): [True: 101k, False: 877k]
  ------------------
  746|   101k|        tx_split[depth] & (1 << (y_off * 4 + x_off)))
  ------------------
  |  Branch (746:9): [True: 78.5k, False: 23.0k]
  ------------------
  747|  78.5k|    {
  748|  78.5k|        const enum RectTxfmSize sub = t_dim->sub;
  749|  78.5k|        const TxfmInfo *const sub_t_dim = &dav1d_txfm_dimensions[sub];
  750|  78.5k|        const int txsw = sub_t_dim->w, txsh = sub_t_dim->h;
  751|       |
  752|  78.5k|        read_coef_tree(t, bs, b, sub, depth + 1, tx_split,
  753|  78.5k|                       x_off * 2 + 0, y_off * 2 + 0, dst);
  754|  78.5k|        t->bx += txsw;
  755|  78.5k|        if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (755:13): [True: 63.0k, False: 15.5k]
  |  Branch (755:27): [True: 62.4k, False: 576]
  ------------------
  756|  62.4k|            read_coef_tree(t, bs, b, sub, depth + 1, tx_split, x_off * 2 + 1,
  757|  62.4k|                           y_off * 2 + 0, dst ? &dst[4 * txsw] : NULL);
  ------------------
  |  Branch (757:43): [True: 62.4k, False: 0]
  ------------------
  758|  78.5k|        t->bx -= txsw;
  759|  78.5k|        t->by += txsh;
  760|  78.5k|        if (txh >= txw && t->by < f->bh) {
  ------------------
  |  Branch (760:13): [True: 55.7k, False: 22.7k]
  |  Branch (760:27): [True: 54.5k, False: 1.25k]
  ------------------
  761|  54.5k|            if (dst)
  ------------------
  |  Branch (761:17): [True: 54.5k, False: 0]
  ------------------
  762|  54.5k|                dst += 4 * txsh * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|  54.5k|#define PXSTRIDE(x) (x)
  ------------------
  763|  54.5k|            read_coef_tree(t, bs, b, sub, depth + 1, tx_split,
  764|  54.5k|                           x_off * 2 + 0, y_off * 2 + 1, dst);
  765|  54.5k|            t->bx += txsw;
  766|  54.5k|            if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (766:17): [True: 39.0k, False: 15.5k]
  |  Branch (766:31): [True: 38.4k, False: 558]
  ------------------
  767|  38.4k|                read_coef_tree(t, bs, b, sub, depth + 1, tx_split, x_off * 2 + 1,
  768|  38.4k|                               y_off * 2 + 1, dst ? &dst[4 * txsw] : NULL);
  ------------------
  |  Branch (768:47): [True: 38.4k, False: 0]
  ------------------
  769|  54.5k|            t->bx -= txsw;
  770|  54.5k|        }
  771|  78.5k|        t->by -= txsh;
  772|   985k|    } else {
  773|   985k|        const int bx4 = t->bx & 31, by4 = t->by & 31;
  774|   985k|        enum TxfmType txtp;
  775|   985k|        uint8_t cf_ctx;
  776|   985k|        int eob;
  777|   985k|        coef *cf;
  778|       |
  779|   985k|        if (t->frame_thread.pass) {
  ------------------
  |  Branch (779:13): [True: 0, False: 985k]
  ------------------
  780|      0|            const int p = t->frame_thread.pass & 1;
  781|      0|            assert(ts->frame_thread[p].cf);
  ------------------
  |  Branch (781:13): [True: 0, False: 0]
  ------------------
  782|      0|            cf = ts->frame_thread[p].cf;
  783|      0|            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
  784|   985k|        } else {
  785|   985k|            cf = bitfn(t->cf);
  ------------------
  |  |   51|   985k|#define bitfn(x) x##_8bpc
  ------------------
  786|   985k|        }
  787|   985k|        if (t->frame_thread.pass != 2) {
  ------------------
  |  Branch (787:13): [True: 985k, False: 0]
  ------------------
  788|   985k|            eob = decode_coefs(t, &t->a->lcoef[bx4], &t->l.lcoef[by4],
  789|   985k|                               ytx, bs, b, 0, 0, cf, &txtp, &cf_ctx);
  790|   985k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   985k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 985k]
  |  |  ------------------
  |  |   35|   985k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   985k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  791|      0|                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
  792|      0|                       ytx, txtp, eob, ts->msac.rng);
  793|   985k|            dav1d_memset_likely_pow2(&t->a->lcoef[bx4], cf_ctx, imin(txw, f->bw - t->bx));
  794|   985k|            dav1d_memset_likely_pow2(&t->l.lcoef[by4], cf_ctx, imin(txh, f->bh - t->by));
  795|   985k|#define set_ctx(rep_macro) \
  796|   985k|            for (int y = 0; y < txh; y++) { \
  797|   985k|                rep_macro(txtp_map, 0, txtp); \
  798|   985k|                txtp_map += 32; \
  799|   985k|            }
  800|   985k|            uint8_t *txtp_map = &t->scratch.txtp_map[by4 * 32 + bx4];
  801|   985k|            case_set_upto16(t_dim->lw);
  ------------------
  |  |   80|   985k|    switch (var) { \
  |  |   81|   506k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  796|  1.07M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 568k, False: 506k]
  |  |  |  |  ------------------
  |  |  |  |  797|   568k|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|   568k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   568k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|   568k|                txtp_map += 32; \
  |  |  |  |  799|   568k|            }
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 506k, False: 479k]
  |  |  ------------------
  |  |   82|   216k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  796|   735k|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 518k, False: 216k]
  |  |  |  |  ------------------
  |  |  |  |  797|   518k|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|   518k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   518k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|   518k|                txtp_map += 32; \
  |  |  |  |  799|   518k|            }
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 216k, False: 768k]
  |  |  ------------------
  |  |   83|   168k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  796|   705k|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 537k, False: 168k]
  |  |  |  |  ------------------
  |  |  |  |  797|   537k|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|   537k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   537k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|   537k|                txtp_map += 32; \
  |  |  |  |  799|   537k|            }
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 168k, False: 817k]
  |  |  ------------------
  |  |   84|  53.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  796|   348k|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 295k, False: 53.3k]
  |  |  |  |  ------------------
  |  |  |  |  797|   295k|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|   295k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   295k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|   295k|                txtp_map += 32; \
  |  |  |  |  799|   295k|            }
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 53.3k, False: 931k]
  |  |  ------------------
  |  |   85|  41.0k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  796|   615k|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 574k, False: 41.0k]
  |  |  |  |  ------------------
  |  |  |  |  797|   574k|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|   574k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   574k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   574k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   574k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 574k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|   574k|                txtp_map += 32; \
  |  |  |  |  799|   574k|            }
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 41.0k, False: 944k]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 985k]
  |  |  ------------------
  |  |   87|   985k|    }
  ------------------
  |  Branch (801:13): [Folded, False: 0]
  ------------------
  802|   985k|#undef set_ctx
  803|   985k|            if (t->frame_thread.pass == 1)
  ------------------
  |  Branch (803:17): [True: 0, False: 985k]
  ------------------
  804|      0|                *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  805|   985k|        } else {
  806|      0|            const int cbi = *ts->frame_thread[0].cbi++;
  807|      0|            eob  = cbi >> 5;
  808|      0|            txtp = cbi & 0x1f;
  809|      0|        }
  810|   985k|        if (!(t->frame_thread.pass & 1)) {
  ------------------
  |  Branch (810:13): [True: 985k, False: 0]
  ------------------
  811|   985k|            assert(dst);
  ------------------
  |  Branch (811:13): [True: 985k, False: 0]
  ------------------
  812|   985k|            if (eob >= 0) {
  ------------------
  |  Branch (812:17): [True: 678k, False: 307k]
  ------------------
  813|   678k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   678k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 678k]
  |  |  ------------------
  |  |   35|   678k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   678k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
  814|      0|                    coef_dump(cf, imin(t_dim->h, 8) * 4, imin(t_dim->w, 8) * 4, 3, "dq");
  815|   678k|                dsp->itx.itxfm_add[ytx][txtp](dst, f->cur.stride[0], cf, eob
  816|   678k|                                              HIGHBD_CALL_SUFFIX);
  817|   678k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   678k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 678k]
  |  |  ------------------
  |  |   35|   678k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   678k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
  818|      0|                    hex_dump(dst, f->cur.stride[0], t_dim->w * 4, t_dim->h * 4, "recon");
  819|   678k|            }
  820|   985k|        }
  821|   985k|    }
  822|  1.06M|}
recon_tmpl.c:decode_coefs:
  327|  8.32M|{
  328|  8.32M|    Dav1dTileState *const ts = t->ts;
  329|  8.32M|    const int chroma = !!plane;
  330|  8.32M|    const Dav1dFrameContext *const f = t->f;
  331|  8.32M|    const int lossless = f->frame_hdr->segmentation.lossless[b->seg_id];
  332|  8.32M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  333|  8.32M|    const int dbg = DEBUG_BLOCK_INFO && plane && 0;
  ------------------
  |  |   34|  8.32M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 8.32M]
  |  |  ------------------
  |  |   35|  8.32M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  8.32M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (333:41): [True: 0, False: 0]
  |  Branch (333:50): [Folded, False: 0]
  ------------------
  334|       |
  335|  8.32M|    if (dbg)
  ------------------
  |  Branch (335:9): [Folded, False: 8.32M]
  ------------------
  336|      0|        printf("Start: r=%d\n", ts->msac.rng);
  337|       |
  338|       |    // does this block have any non-zero coefficients
  339|  8.32M|    const int sctx = get_skip_ctx(t_dim, bs, a, l, chroma, f->cur.p.layout);
  340|  8.32M|    const int all_skip = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  8.32M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  341|  8.32M|                             ts->cdf.coef.skip[t_dim->ctx][sctx]);
  342|  8.32M|    if (dbg)
  ------------------
  |  Branch (342:9): [Folded, False: 8.32M]
  ------------------
  343|      0|        printf("Post-non-zero[%d][%d][%d]: r=%d\n",
  344|      0|               t_dim->ctx, sctx, all_skip, ts->msac.rng);
  345|  8.32M|    if (all_skip) {
  ------------------
  |  Branch (345:9): [True: 4.32M, False: 3.99M]
  ------------------
  346|  4.32M|        *res_ctx = 0x40;
  347|  4.32M|        *txtp = lossless * WHT_WHT; /* lossless ? WHT_WHT : DCT_DCT */
  348|  4.32M|        return -1;
  349|  4.32M|    }
  350|       |
  351|       |    // transform type (chroma: derived, luma: explicitly coded)
  352|  3.99M|    if (lossless) {
  ------------------
  |  Branch (352:9): [True: 643k, False: 3.35M]
  ------------------
  353|   643k|        assert(t_dim->max == TX_4X4);
  ------------------
  |  Branch (353:9): [True: 643k, False: 0]
  ------------------
  354|   643k|        *txtp = WHT_WHT;
  355|  3.35M|    } else if (t_dim->max + intra >= TX_64X64) {
  ------------------
  |  Branch (355:16): [True: 738k, False: 2.61M]
  ------------------
  356|   738k|        *txtp = DCT_DCT;
  357|  2.61M|    } else if (chroma) {
  ------------------
  |  Branch (357:16): [True: 633k, False: 1.98M]
  ------------------
  358|       |        // inferred from either the luma txtp (inter) or a LUT (intra)
  359|   633k|        *txtp = intra ? dav1d_txtp_from_uvmode[b->uv_mode] :
  ------------------
  |  Branch (359:17): [True: 435k, False: 197k]
  ------------------
  360|   633k|                        get_uv_inter_txtp(t_dim, *txtp);
  361|  1.98M|    } else if (!f->frame_hdr->segmentation.qidx[b->seg_id]) {
  ------------------
  |  Branch (361:16): [True: 6.43k, False: 1.97M]
  ------------------
  362|       |        // In libaom, lossless is checked by a literal qidx == 0, but not all
  363|       |        // such blocks are actually lossless. The remainder gets an implicit
  364|       |        // transform type (for luma)
  365|  6.43k|        *txtp = DCT_DCT;
  366|  1.97M|    } else {
  367|  1.97M|        unsigned idx;
  368|  1.97M|        if (intra) {
  ------------------
  |  Branch (368:13): [True: 1.51M, False: 459k]
  ------------------
  369|  1.51M|            const enum IntraPredMode y_mode_nofilt = b->y_mode == FILTER_PRED ?
  ------------------
  |  Branch (369:54): [True: 256k, False: 1.25M]
  ------------------
  370|  1.25M|                dav1d_filter_mode_to_y_mode[b->y_angle] : b->y_mode;
  371|  1.51M|            if (f->frame_hdr->reduced_txtp_set || t_dim->min == TX_16X16) {
  ------------------
  |  Branch (371:17): [True: 264k, False: 1.25M]
  |  Branch (371:51): [True: 211k, False: 1.03M]
  ------------------
  372|   475k|                idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   475k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  373|   475k|                          ts->cdf.m.txtp_intra2[t_dim->min][y_mode_nofilt], 4);
  374|   475k|                *txtp = dav1d_tx_types_per_set[idx + 0];
  375|  1.03M|            } else {
  376|  1.03M|                idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  1.03M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  377|  1.03M|                          ts->cdf.m.txtp_intra1[t_dim->min][y_mode_nofilt], 6);
  378|  1.03M|                *txtp = dav1d_tx_types_per_set[idx + 5];
  379|  1.03M|            }
  380|  1.51M|            if (dbg)
  ------------------
  |  Branch (380:17): [Folded, False: 1.51M]
  ------------------
  381|      0|                printf("Post-txtp-intra[%d->%d][%d][%d->%d]: r=%d\n",
  382|      0|                       tx, t_dim->min, y_mode_nofilt, idx, *txtp, ts->msac.rng);
  383|  1.51M|        } else {
  384|   459k|            if (f->frame_hdr->reduced_txtp_set || t_dim->max == TX_32X32) {
  ------------------
  |  Branch (384:17): [True: 94.2k, False: 365k]
  |  Branch (384:51): [True: 46.7k, False: 318k]
  ------------------
  385|   140k|                idx = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   140k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  386|   140k|                          ts->cdf.m.txtp_inter3[t_dim->min]);
  387|   140k|                *txtp = (idx - 1) & IDTX; /* idx ? DCT_DCT : IDTX */
  388|   318k|            } else if (t_dim->min == TX_16X16) {
  ------------------
  |  Branch (388:24): [True: 53.4k, False: 265k]
  ------------------
  389|  53.4k|                idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  53.4k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
  390|  53.4k|                          ts->cdf.m.txtp_inter2, 11);
  391|  53.4k|                *txtp = dav1d_tx_types_per_set[idx + 12];
  392|   265k|            } else {
  393|   265k|                idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   265k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
  394|   265k|                          ts->cdf.m.txtp_inter1[t_dim->min], 15);
  395|   265k|                *txtp = dav1d_tx_types_per_set[idx + 24];
  396|   265k|            }
  397|   459k|            if (dbg)
  ------------------
  |  Branch (397:17): [Folded, False: 459k]
  ------------------
  398|      0|                printf("Post-txtp-inter[%d->%d][%d->%d]: r=%d\n",
  399|      0|                       tx, t_dim->min, idx, *txtp, ts->msac.rng);
  400|   459k|        }
  401|  1.97M|    }
  402|       |
  403|       |    // find end-of-block (eob)
  404|  3.99M|    int eob;
  405|  3.99M|    const int slw = imin(t_dim->lw, TX_32X32), slh = imin(t_dim->lh, TX_32X32);
  406|  3.99M|    const int tx2dszctx = slw + slh;
  407|  3.99M|    const enum TxClass tx_class = dav1d_tx_type_class[*txtp];
  408|  3.99M|    const int is_1d = tx_class != TX_CLASS_2D;
  409|  3.99M|    switch (tx2dszctx) {
  ------------------
  |  Branch (409:13): [True: 3.99M, False: 0]
  ------------------
  410|      0|#define case_sz(sz, bin, ns, is_1d) \
  411|      0|    case sz: { \
  412|      0|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  413|      0|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  414|      0|        break; \
  415|      0|    }
  416|  1.06M|    case_sz(0,   16,  8, [is_1d]);
  ------------------
  |  |  411|  1.06M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 1.06M, False: 2.93M]
  |  |  ------------------
  |  |  412|  1.06M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  1.06M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|  1.06M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|  1.06M|        break; \
  |  |  415|  1.06M|    }
  ------------------
  417|   354k|    case_sz(1,   32,  8, [is_1d]);
  ------------------
  |  |  411|   354k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 354k, False: 3.64M]
  |  |  ------------------
  |  |  412|   354k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   354k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|   354k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|   354k|        break; \
  |  |  415|   354k|    }
  ------------------
  418|   895k|    case_sz(2,   64,  8, [is_1d]);
  ------------------
  |  |  411|   895k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 895k, False: 3.10M]
  |  |  ------------------
  |  |  412|   895k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   895k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|   895k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|   895k|        break; \
  |  |  415|   895k|    }
  ------------------
  419|   443k|    case_sz(3,  128,  8, [is_1d]);
  ------------------
  |  |  411|   443k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 443k, False: 3.55M]
  |  |  ------------------
  |  |  412|   443k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   443k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|   443k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|   443k|        break; \
  |  |  415|   443k|    }
  ------------------
  420|   551k|    case_sz(4,  256, 16, [is_1d]);
  ------------------
  |  |  411|   551k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 551k, False: 3.44M]
  |  |  ------------------
  |  |  412|   551k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   551k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|   551k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|   551k|        break; \
  |  |  415|   551k|    }
  ------------------
  421|   225k|    case_sz(5,  512, 16,        );
  ------------------
  |  |  411|   225k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 225k, False: 3.77M]
  |  |  ------------------
  |  |  412|   225k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   225k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|   225k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|   225k|        break; \
  |  |  415|   225k|    }
  ------------------
  422|   463k|    case_sz(6, 1024, 16,        );
  ------------------
  |  |  411|   463k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 463k, False: 3.53M]
  |  |  ------------------
  |  |  412|   463k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   463k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|   463k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|   463k|        break; \
  |  |  415|   463k|    }
  ------------------
  423|  3.99M|#undef case_sz
  424|  3.99M|    }
  425|  3.99M|    if (dbg)
  ------------------
  |  Branch (425:9): [Folded, False: 3.99M]
  ------------------
  426|      0|        printf("Post-eob_bin_%d[%d][%d][%d]: r=%d\n",
  427|      0|               16 << tx2dszctx, chroma, is_1d, eob, ts->msac.rng);
  428|  3.99M|    if (eob > 1) {
  ------------------
  |  Branch (428:9): [True: 2.81M, False: 1.18M]
  ------------------
  429|  2.81M|        const int eob_bin = eob - 2;
  430|  2.81M|        uint16_t *const eob_hi_bit_cdf =
  431|  2.81M|            ts->cdf.coef.eob_hi_bit[t_dim->ctx][chroma][eob_bin];
  432|  2.81M|        const int eob_hi_bit = dav1d_msac_decode_bool_adapt(&ts->msac, eob_hi_bit_cdf);
  ------------------
  |  |   52|  2.81M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  433|  2.81M|        if (dbg)
  ------------------
  |  Branch (433:13): [Folded, False: 2.81M]
  ------------------
  434|      0|            printf("Post-eob_hi_bit[%d][%d][%d][%d]: r=%d\n",
  435|      0|                   t_dim->ctx, chroma, eob_bin, eob_hi_bit, ts->msac.rng);
  436|  2.81M|        eob = ((eob_hi_bit | 2) << eob_bin) | dav1d_msac_decode_bools(&ts->msac, eob_bin);
  437|  2.81M|        if (dbg)
  ------------------
  |  Branch (437:13): [Folded, False: 2.81M]
  ------------------
  438|      0|            printf("Post-eob[%d]: r=%d\n", eob, ts->msac.rng);
  439|  2.81M|    }
  440|  3.99M|    assert(eob >= 0);
  ------------------
  |  Branch (440:5): [True: 3.99M, False: 0]
  ------------------
  441|       |
  442|       |    // base tokens
  443|  3.99M|    uint16_t (*const eob_cdf)[4] = ts->cdf.coef.eob_base_tok[t_dim->ctx][chroma];
  444|  3.99M|    uint16_t (*const hi_cdf)[4] = ts->cdf.coef.br_tok[imin(t_dim->ctx, 3)][chroma];
  445|  3.99M|    unsigned rc, dc_tok;
  446|       |
  447|  3.99M|    if (eob) {
  ------------------
  |  Branch (447:9): [True: 2.96M, False: 1.03M]
  ------------------
  448|  2.96M|        uint16_t (*const lo_cdf)[4] = ts->cdf.coef.base_tok[t_dim->ctx][chroma];
  449|  2.96M|        uint8_t *const levels = t->scratch.levels; // bits 0-5: tok, 6-7: lo_tok
  450|       |
  451|       |        /* eob */
  452|  2.96M|        unsigned ctx = 1 + (eob > 2 << tx2dszctx) + (eob > 4 << tx2dszctx);
  453|  2.96M|        int eob_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, eob_cdf[ctx], 2);
  ------------------
  |  |   47|  2.96M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  454|  2.96M|        int tok = eob_tok + 1;
  455|  2.96M|        int level_tok = tok * 0x41;
  456|  2.96M|        unsigned mag;
  457|       |
  458|  2.96M|#define DECODE_COEFS_CLASS(tx_class) \
  459|  2.96M|        unsigned x, y; \
  460|  2.96M|        uint8_t *level; \
  461|  2.96M|        if (tx_class == TX_CLASS_2D) \
  462|  2.96M|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  463|  2.96M|        else if (tx_class == TX_CLASS_H) \
  464|       |            /* Transposing reduces the stride and padding requirements */ \
  465|  2.96M|            x = eob & mask, y = eob >> shift, rc = eob; \
  466|  2.96M|        else /* tx_class == TX_CLASS_V */ \
  467|  2.96M|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  468|  2.96M|        if (dbg) \
  469|  2.96M|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  470|  2.96M|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  471|  2.96M|        if (eob_tok == 2) { \
  472|  2.96M|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  473|  2.96M|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  474|  2.96M|            level_tok = tok + (3 << 6); \
  475|  2.96M|            if (dbg) \
  476|  2.96M|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  477|  2.96M|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  478|  2.96M|                       ts->msac.rng); \
  479|  2.96M|        } \
  480|  2.96M|        cf[rc] = tok << 11; \
  481|  2.96M|        if (tx_class == TX_CLASS_2D) \
  482|  2.96M|            level = levels + rc; \
  483|  2.96M|        else \
  484|  2.96M|            level = levels + x * stride + y; \
  485|  2.96M|        *level = (uint8_t) level_tok; \
  486|  2.96M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  487|  2.96M|            unsigned rc_i; \
  488|  2.96M|            if (tx_class == TX_CLASS_2D) \
  489|  2.96M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  490|  2.96M|            else if (tx_class == TX_CLASS_H) \
  491|  2.96M|                x = i & mask, y = i >> shift, rc_i = i; \
  492|  2.96M|            else /* tx_class == TX_CLASS_V */ \
  493|  2.96M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  494|  2.96M|            assert(x < 32 && y < 32); \
  495|  2.96M|            if (tx_class == TX_CLASS_2D) \
  496|  2.96M|                level = levels + rc_i; \
  497|  2.96M|            else \
  498|  2.96M|                level = levels + x * stride + y; \
  499|  2.96M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  500|  2.96M|            if (tx_class == TX_CLASS_2D) \
  501|  2.96M|                y |= x; \
  502|  2.96M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  503|  2.96M|            if (dbg) \
  504|  2.96M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  505|  2.96M|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  506|  2.96M|            if (tok == 3) { \
  507|  2.96M|                mag &= 63; \
  508|  2.96M|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  509|  2.96M|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  510|  2.96M|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  511|  2.96M|                if (dbg) \
  512|  2.96M|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  513|  2.96M|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  514|  2.96M|                           ts->msac.rng); \
  515|  2.96M|                *level = (uint8_t) (tok + (3 << 6)); \
  516|  2.96M|                cf[rc_i] = (tok << 11) | rc; \
  517|  2.96M|                rc = rc_i; \
  518|  2.96M|            } else { \
  519|       |                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  520|  2.96M|                tok *= 0x17ff41; \
  521|  2.96M|                *level = (uint8_t) tok; \
  522|       |                /* tok ? (tok << 11) | rc : 0 */ \
  523|  2.96M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  524|  2.96M|                if (tok) rc = rc_i; \
  525|  2.96M|                cf[rc_i] = tok; \
  526|  2.96M|            } \
  527|  2.96M|        } \
  528|       |        /* dc */ \
  529|  2.96M|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  530|  2.96M|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  531|  2.96M|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  532|  2.96M|        if (dbg) \
  533|  2.96M|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  534|  2.96M|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  535|  2.96M|        if (dc_tok == 3) { \
  536|  2.96M|            if (tx_class == TX_CLASS_2D) \
  537|  2.96M|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  538|  2.96M|                      levels[1 * stride + 1]; \
  539|  2.96M|            mag &= 63; \
  540|  2.96M|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  541|  2.96M|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  542|  2.96M|            if (dbg) \
  543|  2.96M|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  544|  2.96M|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  545|  2.96M|        } \
  546|  2.96M|        break
  547|       |
  548|  2.96M|        const uint16_t *scan;
  549|  2.96M|        switch (tx_class) {
  550|  2.71M|        case TX_CLASS_2D: {
  ------------------
  |  Branch (550:9): [True: 2.71M, False: 247k]
  ------------------
  551|  2.71M|            const unsigned nonsquare_tx = tx >= RTX_4X8;
  552|  2.71M|            const uint8_t (*const lo_ctx_offsets)[5] =
  553|  2.71M|                dav1d_lo_ctx_offsets[nonsquare_tx + (tx & nonsquare_tx)];
  554|  2.71M|            scan = dav1d_scans[tx];
  555|  2.71M|            const ptrdiff_t stride = 4 << slh;
  556|  2.71M|            const unsigned shift = slh + 2, shift2 = 0;
  557|  2.71M|            const unsigned mask = (4 << slh) - 1;
  558|  2.71M|            memset(levels, 0, stride * ((4 << slw) + 2));
  559|  2.71M|            DECODE_COEFS_CLASS(TX_CLASS_2D);
  ------------------
  |  |  459|  2.71M|        unsigned x, y; \
  |  |  460|  2.71M|        uint8_t *level; \
  |  |  461|  2.71M|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [True: 2.71M, Folded]
  |  |  ------------------
  |  |  462|  2.71M|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|  2.71M|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [Folded, False: 0]
  |  |  ------------------
  |  |  464|      0|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|      0|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|      0|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|      0|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|  2.71M|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 2.71M]
  |  |  ------------------
  |  |  469|  2.71M|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|  2.71M|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 100k, False: 2.61M]
  |  |  ------------------
  |  |  472|   100k|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 96.5k, False: 3.71k]
  |  |  |  Branch (472:20): [True: 100k, Folded]
  |  |  ------------------
  |  |  473|   100k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|   100k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|   100k|            level_tok = tok + (3 << 6); \
  |  |  475|   100k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 100k]
  |  |  ------------------
  |  |  476|   100k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|   100k|        } \
  |  |  480|  2.71M|        cf[rc] = tok << 11; \
  |  |  481|  2.71M|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [True: 2.71M, Folded]
  |  |  ------------------
  |  |  482|  2.71M|            level = levels + rc; \
  |  |  483|  2.71M|        else \
  |  |  484|  2.71M|            level = levels + x * stride + y; \
  |  |  485|  2.71M|        *level = (uint8_t) level_tok; \
  |  |  486|  78.7M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 76.0M, False: 2.71M]
  |  |  ------------------
  |  |  487|  76.0M|            unsigned rc_i; \
  |  |  488|  76.0M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [True: 76.0M, Folded]
  |  |  ------------------
  |  |  489|  76.0M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|  76.0M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [Folded, False: 0]
  |  |  ------------------
  |  |  491|      0|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|      0|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|      0|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|  76.0M|            assert(x < 32 && y < 32); \
  |  |  495|  76.0M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [True: 76.0M, Folded]
  |  |  ------------------
  |  |  496|  76.0M|                level = levels + rc_i; \
  |  |  497|  76.0M|            else \
  |  |  498|  76.0M|                level = levels + x * stride + y; \
  |  |  499|  76.0M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|  76.0M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [True: 76.0M, Folded]
  |  |  ------------------
  |  |  501|  76.0M|                y |= x; \
  |  |  502|  76.0M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  76.0M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|  76.0M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 76.0M]
  |  |  ------------------
  |  |  504|  76.0M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|  76.0M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 6.51M, False: 69.5M]
  |  |  ------------------
  |  |  507|  6.51M|                mag &= 63; \
  |  |  508|  6.51M|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 5.35M, False: 1.16M]
  |  |  ------------------
  |  |  509|  6.51M|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 1.83M, False: 4.67M]
  |  |  ------------------
  |  |  510|  6.51M|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  6.51M|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|  6.51M|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 6.51M]
  |  |  ------------------
  |  |  512|  6.51M|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|  6.51M|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|  6.51M|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|  6.51M|                rc = rc_i; \
  |  |  518|  69.5M|            } else { \
  |  |  519|  69.5M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|  69.5M|                tok *= 0x17ff41; \
  |  |  521|  69.5M|                *level = (uint8_t) tok; \
  |  |  522|  69.5M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|  69.5M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|  69.5M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 19.9M, False: 49.5M]
  |  |  ------------------
  |  |  525|  69.5M|                cf[rc_i] = tok; \
  |  |  526|  69.5M|            } \
  |  |  527|  76.0M|        } \
  |  |  528|  2.71M|        /* dc */ \
  |  |  529|  2.71M|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [True: 2.71M, Folded]
  |  |  ------------------
  |  |  530|  2.71M|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|  2.71M|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  2.71M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|  2.71M|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 2.71M]
  |  |  ------------------
  |  |  533|  2.71M|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|  2.71M|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 1.05M, False: 1.66M]
  |  |  ------------------
  |  |  536|  1.05M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [True: 1.05M, Folded]
  |  |  ------------------
  |  |  537|  1.05M|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|  1.05M|                      levels[1 * stride + 1]; \
  |  |  539|  1.05M|            mag &= 63; \
  |  |  540|  1.05M|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 146k, False: 910k]
  |  |  ------------------
  |  |  541|  1.05M|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  1.05M|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  1.05M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 1.05M]
  |  |  ------------------
  |  |  543|  1.05M|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  1.05M|        } \
  |  |  546|  2.71M|        break
  ------------------
  |  Branch (559:13): [True: 76.0M, False: 0]
  |  Branch (559:13): [True: 76.0M, False: 0]
  ------------------
  560|  2.71M|        }
  561|   157k|        case TX_CLASS_H: {
  ------------------
  |  Branch (561:9): [True: 157k, False: 2.80M]
  ------------------
  562|   157k|            const uint8_t (*const lo_ctx_offsets)[5] = NULL;
  563|   157k|            const ptrdiff_t stride = 16;
  564|   157k|            const unsigned shift = slh + 2, shift2 = 0;
  565|   157k|            const unsigned mask = (4 << slh) - 1;
  566|   157k|            memset(levels, 0, stride * ((4 << slh) + 2));
  567|   157k|            DECODE_COEFS_CLASS(TX_CLASS_H);
  ------------------
  |  |  459|   157k|        unsigned x, y; \
  |  |  460|   157k|        uint8_t *level; \
  |  |  461|   157k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [Folded, False: 157k]
  |  |  ------------------
  |  |  462|   157k|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|   157k|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [True: 157k, Folded]
  |  |  ------------------
  |  |  464|   157k|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|   157k|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|   157k|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|   157k|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|   157k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 157k]
  |  |  ------------------
  |  |  469|   157k|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|   157k|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 3.10k, False: 154k]
  |  |  ------------------
  |  |  472|  3.10k|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 2.70k, False: 398]
  |  |  |  Branch (472:20): [Folded, False: 3.10k]
  |  |  ------------------
  |  |  473|  3.10k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  3.10k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|  3.10k|            level_tok = tok + (3 << 6); \
  |  |  475|  3.10k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 3.10k]
  |  |  ------------------
  |  |  476|  3.10k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|  3.10k|        } \
  |  |  480|   157k|        cf[rc] = tok << 11; \
  |  |  481|   157k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [Folded, False: 157k]
  |  |  ------------------
  |  |  482|   157k|            level = levels + rc; \
  |  |  483|   157k|        else \
  |  |  484|   157k|            level = levels + x * stride + y; \
  |  |  485|   157k|        *level = (uint8_t) level_tok; \
  |  |  486|  3.16M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 3.00M, False: 157k]
  |  |  ------------------
  |  |  487|  3.00M|            unsigned rc_i; \
  |  |  488|  3.00M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [Folded, False: 3.00M]
  |  |  ------------------
  |  |  489|  3.00M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|  3.00M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [True: 3.00M, Folded]
  |  |  ------------------
  |  |  491|  3.00M|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|  3.00M|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|  3.00M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|  3.00M|            assert(x < 32 && y < 32); \
  |  |  495|  3.00M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [Folded, False: 3.00M]
  |  |  ------------------
  |  |  496|  3.00M|                level = levels + rc_i; \
  |  |  497|  3.00M|            else \
  |  |  498|  3.00M|                level = levels + x * stride + y; \
  |  |  499|  3.00M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|  3.00M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [Folded, False: 3.00M]
  |  |  ------------------
  |  |  501|  3.00M|                y |= x; \
  |  |  502|  3.00M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  3.00M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|  3.00M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 3.00M]
  |  |  ------------------
  |  |  504|  3.00M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|  3.00M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 132k, False: 2.87M]
  |  |  ------------------
  |  |  507|   132k|                mag &= 63; \
  |  |  508|   132k|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 79.5k, False: 52.9k]
  |  |  ------------------
  |  |  509|   132k|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 11.3k, False: 121k]
  |  |  ------------------
  |  |  510|   132k|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|   132k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|   132k|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 132k]
  |  |  ------------------
  |  |  512|   132k|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|   132k|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|   132k|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|   132k|                rc = rc_i; \
  |  |  518|  2.87M|            } else { \
  |  |  519|  2.87M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|  2.87M|                tok *= 0x17ff41; \
  |  |  521|  2.87M|                *level = (uint8_t) tok; \
  |  |  522|  2.87M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|  2.87M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|  2.87M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 755k, False: 2.12M]
  |  |  ------------------
  |  |  525|  2.87M|                cf[rc_i] = tok; \
  |  |  526|  2.87M|            } \
  |  |  527|  3.00M|        } \
  |  |  528|   157k|        /* dc */ \
  |  |  529|   157k|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [Folded, False: 157k]
  |  |  ------------------
  |  |  530|   157k|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|   157k|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|   157k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|   157k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 157k]
  |  |  ------------------
  |  |  533|   157k|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|   157k|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 19.1k, False: 138k]
  |  |  ------------------
  |  |  536|  19.1k|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [Folded, False: 19.1k]
  |  |  ------------------
  |  |  537|  19.1k|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|      0|                      levels[1 * stride + 1]; \
  |  |  539|  19.1k|            mag &= 63; \
  |  |  540|  19.1k|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 3.10k, False: 16.0k]
  |  |  ------------------
  |  |  541|  19.1k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  19.1k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  19.1k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 19.1k]
  |  |  ------------------
  |  |  543|  19.1k|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  19.1k|        } \
  |  |  546|   157k|        break
  ------------------
  |  Branch (567:13): [True: 3.00M, False: 0]
  |  Branch (567:13): [True: 3.00M, False: 0]
  ------------------
  568|   157k|        }
  569|  90.5k|        case TX_CLASS_V: {
  ------------------
  |  Branch (569:9): [True: 90.5k, False: 2.87M]
  ------------------
  570|  90.5k|            const uint8_t (*const lo_ctx_offsets)[5] = NULL;
  571|  90.5k|            const ptrdiff_t stride = 16;
  572|  90.5k|            const unsigned shift = slw + 2, shift2 = slh + 2;
  573|  90.5k|            const unsigned mask = (4 << slw) - 1;
  574|  90.5k|            memset(levels, 0, stride * ((4 << slw) + 2));
  575|  90.5k|            DECODE_COEFS_CLASS(TX_CLASS_V);
  ------------------
  |  |  459|  90.5k|        unsigned x, y; \
  |  |  460|  90.5k|        uint8_t *level; \
  |  |  461|  90.5k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  462|  90.5k|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|  90.5k|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  464|  90.5k|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|  90.5k|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|  90.5k|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|  90.5k|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|  90.5k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  469|  90.5k|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|  90.5k|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 3.33k, False: 87.2k]
  |  |  ------------------
  |  |  472|  3.33k|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 2.99k, False: 335]
  |  |  |  Branch (472:20): [Folded, False: 3.33k]
  |  |  ------------------
  |  |  473|  3.33k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  3.33k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|  3.33k|            level_tok = tok + (3 << 6); \
  |  |  475|  3.33k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 3.33k]
  |  |  ------------------
  |  |  476|  3.33k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|  3.33k|        } \
  |  |  480|  90.5k|        cf[rc] = tok << 11; \
  |  |  481|  90.5k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  482|  90.5k|            level = levels + rc; \
  |  |  483|  90.5k|        else \
  |  |  484|  90.5k|            level = levels + x * stride + y; \
  |  |  485|  90.5k|        *level = (uint8_t) level_tok; \
  |  |  486|  1.60M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 1.51M, False: 90.5k]
  |  |  ------------------
  |  |  487|  1.51M|            unsigned rc_i; \
  |  |  488|  1.51M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [Folded, False: 1.51M]
  |  |  ------------------
  |  |  489|  1.51M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|  1.51M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [Folded, False: 1.51M]
  |  |  ------------------
  |  |  491|  1.51M|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|  1.51M|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|  1.51M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|  1.51M|            assert(x < 32 && y < 32); \
  |  |  495|  1.51M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [Folded, False: 1.51M]
  |  |  ------------------
  |  |  496|  1.51M|                level = levels + rc_i; \
  |  |  497|  1.51M|            else \
  |  |  498|  1.51M|                level = levels + x * stride + y; \
  |  |  499|  1.51M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|  1.51M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [Folded, False: 1.51M]
  |  |  ------------------
  |  |  501|  1.51M|                y |= x; \
  |  |  502|  1.51M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  1.51M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|  1.51M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 1.51M]
  |  |  ------------------
  |  |  504|  1.51M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|  1.51M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 65.3k, False: 1.44M]
  |  |  ------------------
  |  |  507|  65.3k|                mag &= 63; \
  |  |  508|  65.3k|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 38.8k, False: 26.4k]
  |  |  ------------------
  |  |  509|  65.3k|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 5.94k, False: 59.4k]
  |  |  ------------------
  |  |  510|  65.3k|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  65.3k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|  65.3k|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 65.3k]
  |  |  ------------------
  |  |  512|  65.3k|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|  65.3k|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|  65.3k|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|  65.3k|                rc = rc_i; \
  |  |  518|  1.44M|            } else { \
  |  |  519|  1.44M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|  1.44M|                tok *= 0x17ff41; \
  |  |  521|  1.44M|                *level = (uint8_t) tok; \
  |  |  522|  1.44M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|  1.44M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|  1.44M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 355k, False: 1.08M]
  |  |  ------------------
  |  |  525|  1.44M|                cf[rc_i] = tok; \
  |  |  526|  1.44M|            } \
  |  |  527|  1.51M|        } \
  |  |  528|  90.5k|        /* dc */ \
  |  |  529|  90.5k|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  530|  90.5k|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|  90.5k|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  90.5k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|  90.5k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 90.5k]
  |  |  ------------------
  |  |  533|  90.5k|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|  90.5k|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 10.2k, False: 80.2k]
  |  |  ------------------
  |  |  536|  10.2k|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [Folded, False: 10.2k]
  |  |  ------------------
  |  |  537|  10.2k|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|      0|                      levels[1 * stride + 1]; \
  |  |  539|  10.2k|            mag &= 63; \
  |  |  540|  10.2k|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 1.97k, False: 8.30k]
  |  |  ------------------
  |  |  541|  10.2k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  10.2k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  10.2k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 10.2k]
  |  |  ------------------
  |  |  543|  10.2k|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  10.2k|        } \
  |  |  546|  90.5k|        break
  ------------------
  |  Branch (575:13): [True: 1.51M, False: 0]
  |  Branch (575:13): [True: 1.51M, False: 0]
  ------------------
  576|  90.5k|        }
  577|      0|#undef DECODE_COEFS_CLASS
  578|      0|        default: assert(0);
  ------------------
  |  Branch (578:9): [True: 0, False: 2.96M]
  |  Branch (578:18): [Folded, False: 0]
  ------------------
  579|  2.96M|        }
  580|  2.96M|    } else { // dc-only
  581|  1.03M|        int tok_br = dav1d_msac_decode_symbol_adapt4(&ts->msac, eob_cdf[0], 2);
  ------------------
  |  |   47|  1.03M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  582|  1.03M|        dc_tok = 1 + tok_br;
  583|  1.03M|        if (dbg)
  ------------------
  |  Branch (583:13): [Folded, False: 1.03M]
  ------------------
  584|      0|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n",
  585|      0|                   t_dim->ctx, chroma, 0, dc_tok, ts->msac.rng);
  586|  1.03M|        if (tok_br == 2) {
  ------------------
  |  Branch (586:13): [True: 69.5k, False: 963k]
  ------------------
  587|  69.5k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[0]);
  ------------------
  |  |   49|  69.5k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  ------------------
  588|  69.5k|            if (dbg)
  ------------------
  |  Branch (588:17): [Folded, False: 69.5k]
  ------------------
  589|      0|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n",
  590|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng);
  591|  69.5k|        }
  592|  1.03M|        rc = 0;
  593|  1.03M|    }
  594|       |
  595|       |    // residual and sign
  596|  3.99M|    const uint16_t *const dq_tbl = ts->dq[b->seg_id][plane];
  597|  3.99M|    const uint8_t *const qm_tbl = *txtp < IDTX ? f->qm[tx][plane] : NULL;
  ------------------
  |  Branch (597:35): [True: 2.91M, False: 1.08M]
  ------------------
  598|  3.99M|    const int dq_shift = imax(0, t_dim->ctx - 2);
  599|  3.99M|    const int cf_max = ~(~127U << (BITDEPTH == 8 ? 8 : f->cur.p.bpc));
  ------------------
  |  Branch (599:36): [True: 1.76M, Folded]
  ------------------
  600|  3.99M|    unsigned cul_level, dc_sign_level;
  601|       |
  602|  3.99M|    if (!dc_tok) {
  ------------------
  |  Branch (602:9): [True: 736k, False: 3.26M]
  ------------------
  603|   736k|        cul_level = 0;
  604|   736k|        dc_sign_level = 1 << 6;
  605|   736k|        if (qm_tbl) goto ac_qm;
  ------------------
  |  Branch (605:13): [True: 96.1k, False: 640k]
  ------------------
  606|   640k|        goto ac_noqm;
  607|   736k|    }
  608|       |
  609|  3.26M|    const int dc_sign_ctx = get_dc_sign_ctx(tx, a, l);
  610|  3.26M|    uint16_t *const dc_sign_cdf = ts->cdf.coef.dc_sign[chroma][dc_sign_ctx];
  611|  3.26M|    const int dc_sign = dav1d_msac_decode_bool_adapt(&ts->msac, dc_sign_cdf);
  ------------------
  |  |   52|  3.26M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  612|  3.26M|    if (dbg)
  ------------------
  |  Branch (612:9): [Folded, False: 3.26M]
  ------------------
  613|      0|        printf("Post-dc_sign[%d][%d][%d]: r=%d\n",
  614|      0|               chroma, dc_sign_ctx, dc_sign, ts->msac.rng);
  615|       |
  616|  3.26M|    int dc_dq = dq_tbl[0];
  617|  3.26M|    dc_sign_level = (dc_sign - 1) & (2 << 6);
  618|       |
  619|  3.26M|    if (qm_tbl) {
  ------------------
  |  Branch (619:9): [True: 531k, False: 2.72M]
  ------------------
  620|   531k|        dc_dq = (dc_dq * qm_tbl[0] + 16) >> 5;
  621|       |
  622|   531k|        if (dc_tok == 15) {
  ------------------
  |  Branch (622:13): [True: 36.2k, False: 495k]
  ------------------
  623|  36.2k|            dc_tok = read_golomb(&ts->msac) + 15;
  624|  36.2k|            if (dbg)
  ------------------
  |  Branch (624:17): [Folded, False: 36.2k]
  ------------------
  625|      0|                printf("Post-dc_residual[%d->%d]: r=%d\n",
  626|      0|                       dc_tok - 15, dc_tok, ts->msac.rng);
  627|       |
  628|  36.2k|            dc_tok &= 0xfffff;
  629|  36.2k|            dc_dq = (dc_dq * dc_tok) & 0xffffff;
  630|   495k|        } else {
  631|   495k|            dc_dq *= dc_tok;
  632|   495k|            assert(dc_dq <= 0xffffff);
  ------------------
  |  Branch (632:13): [True: 495k, False: 0]
  ------------------
  633|   495k|        }
  634|   531k|        cul_level = dc_tok;
  635|   531k|        dc_dq >>= dq_shift;
  636|   531k|        dc_dq = umin(dc_dq, cf_max + dc_sign);
  637|   531k|        cf[0] = (coef) (dc_sign ? -dc_dq : dc_dq);
  ------------------
  |  Branch (637:25): [True: 227k, False: 304k]
  ------------------
  638|       |
  639|   883k|        if (rc) ac_qm: {
  ------------------
  |  Branch (639:13): [True: 393k, False: 138k]
  ------------------
  640|   883k|            const unsigned ac_dq = dq_tbl[1];
  641|  9.17M|            do {
  642|  9.17M|                const int sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  9.17M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  643|  9.17M|                if (dbg)
  ------------------
  |  Branch (643:21): [Folded, False: 9.17M]
  ------------------
  644|      0|                    printf("Post-sign[%d=%d]: r=%d\n", rc, sign, ts->msac.rng);
  645|  9.17M|                const unsigned rc_tok = cf[rc];
  646|  9.17M|                unsigned tok, dq = (ac_dq * qm_tbl[rc] + 16) >> 5;
  647|  9.17M|                int dq_sat;
  648|       |
  649|  9.17M|                if (rc_tok >= (15 << 11)) {
  ------------------
  |  Branch (649:21): [True: 591k, False: 8.58M]
  ------------------
  650|   591k|                    tok = read_golomb(&ts->msac) + 15;
  651|   591k|                    if (dbg)
  ------------------
  |  Branch (651:25): [Folded, False: 591k]
  ------------------
  652|      0|                        printf("Post-residual[%d=%d->%d]: r=%d\n",
  653|      0|                               rc, tok - 15, tok, ts->msac.rng);
  654|       |
  655|   591k|                    tok &= 0xfffff;
  656|   591k|                    dq = (dq * tok) & 0xffffff;
  657|  8.58M|                } else {
  658|  8.58M|                    tok = rc_tok >> 11;
  659|  8.58M|                    dq *= tok;
  660|  8.58M|                    assert(dq <= 0xffffff);
  ------------------
  |  Branch (660:21): [True: 8.58M, False: 0]
  ------------------
  661|  8.58M|                }
  662|  9.17M|                cul_level += tok;
  663|  9.17M|                dq >>= dq_shift;
  664|  9.17M|                dq_sat = umin(dq, cf_max + sign);
  665|  9.17M|                cf[rc] = (coef) (sign ? -dq_sat : dq_sat);
  ------------------
  |  Branch (665:34): [True: 4.74M, False: 4.42M]
  ------------------
  666|       |
  667|  9.17M|                rc = rc_tok & 0x3ff;
  668|  9.17M|            } while (rc);
  ------------------
  |  Branch (668:22): [True: 8.68M, False: 489k]
  ------------------
  669|   883k|        }
  670|  2.72M|    } else {
  671|       |        // non-qmatrix is the common case and allows for additional optimizations
  672|  2.72M|        if (dc_tok == 15) {
  ------------------
  |  Branch (672:13): [True: 103k, False: 2.62M]
  ------------------
  673|   103k|            dc_tok = read_golomb(&ts->msac) + 15;
  674|   103k|            if (dbg)
  ------------------
  |  Branch (674:17): [Folded, False: 103k]
  ------------------
  675|      0|                printf("Post-dc_residual[%d->%d]: r=%d\n",
  676|      0|                       dc_tok - 15, dc_tok, ts->msac.rng);
  677|       |
  678|   103k|            dc_tok &= 0xfffff;
  679|   103k|            dc_dq = ((dc_dq * dc_tok) & 0xffffff) >> dq_shift;
  680|   103k|            dc_dq = umin(dc_dq, cf_max + dc_sign);
  681|  2.62M|        } else {
  682|  2.62M|            dc_dq = ((dc_dq * dc_tok) >> dq_shift);
  683|  2.62M|            assert(dc_dq <= cf_max);
  ------------------
  |  Branch (683:13): [True: 2.62M, False: 0]
  ------------------
  684|  2.62M|        }
  685|  2.72M|        cul_level = dc_tok;
  686|  2.72M|        cf[0] = (coef) (dc_sign ? -dc_dq : dc_dq);
  ------------------
  |  Branch (686:25): [True: 1.34M, False: 1.38M]
  ------------------
  687|       |
  688|  4.30M|        if (rc) ac_noqm: {
  ------------------
  |  Branch (688:13): [True: 1.83M, False: 894k]
  ------------------
  689|  4.30M|            const unsigned ac_dq = dq_tbl[1];
  690|  21.5M|            do {
  691|  21.5M|                const int sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  21.5M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  692|  21.5M|                if (dbg)
  ------------------
  |  Branch (692:21): [Folded, False: 21.5M]
  ------------------
  693|      0|                    printf("Post-sign[%d=%d]: r=%d\n", rc, sign, ts->msac.rng);
  694|  21.5M|                const unsigned rc_tok = cf[rc];
  695|  21.5M|                unsigned tok;
  696|  21.5M|                int dq;
  697|       |
  698|       |                // residual
  699|  21.5M|                if (rc_tok >= (15 << 11)) {
  ------------------
  |  Branch (699:21): [True: 729k, False: 20.8M]
  ------------------
  700|   729k|                    tok = read_golomb(&ts->msac) + 15;
  701|   729k|                    if (dbg)
  ------------------
  |  Branch (701:25): [Folded, False: 729k]
  ------------------
  702|      0|                        printf("Post-residual[%d=%d->%d]: r=%d\n",
  703|      0|                               rc, tok - 15, tok, ts->msac.rng);
  704|       |
  705|       |                    // coefficient parsing, see 5.11.39
  706|   729k|                    tok &= 0xfffff;
  707|       |
  708|       |                    // dequant, see 7.12.3
  709|   729k|                    dq = ((ac_dq * tok) & 0xffffff) >> dq_shift;
  710|   729k|                    dq = umin(dq, cf_max + sign);
  711|  20.8M|                } else {
  712|       |                    // cannot exceed cf_max, so we can avoid the clipping
  713|  20.8M|                    tok = rc_tok >> 11;
  714|  20.8M|                    dq = ((ac_dq * tok) >> dq_shift);
  715|  20.8M|                    assert(dq <= cf_max);
  ------------------
  |  Branch (715:21): [True: 20.8M, False: 0]
  ------------------
  716|  20.8M|                }
  717|  21.5M|                cul_level += tok;
  718|  21.5M|                cf[rc] = (coef) (sign ? -dq : dq);
  ------------------
  |  Branch (718:34): [True: 11.0M, False: 10.5M]
  ------------------
  719|       |
  720|  21.5M|                rc = rc_tok & 0x3ff; // next non-zero rc, zero if eob
  721|  21.5M|            } while (rc);
  ------------------
  |  Branch (721:22): [True: 19.1M, False: 2.47M]
  ------------------
  722|  4.30M|        }
  723|  2.72M|    }
  724|       |
  725|       |    // context
  726|  3.99M|    *res_ctx = umin(cul_level, 63) | dc_sign_level;
  727|       |
  728|  3.99M|    return eob;
  729|  3.26M|}
recon_tmpl.c:get_skip_ctx:
   65|  8.32M|{
   66|  8.32M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
   67|       |
   68|  8.32M|    if (chroma) {
  ------------------
  |  Branch (68:9): [True: 4.17M, False: 4.14M]
  ------------------
   69|  4.17M|        const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
   70|  4.17M|        const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
   71|  4.17M|        const int not_one_blk = b_dim[2] - (!!b_dim[2] && ss_hor) > t_dim->lw ||
  ------------------
  |  Branch (71:33): [True: 1.16M, False: 3.00M]
  |  Branch (71:45): [True: 3.74M, False: 423k]
  |  Branch (71:59): [True: 662k, False: 3.08M]
  ------------------
   72|  3.00M|                                b_dim[3] - (!!b_dim[3] && ss_ver) > t_dim->lh;
  ------------------
  |  Branch (72:33): [True: 81.9k, False: 2.92M]
  |  Branch (72:45): [True: 2.33M, False: 668k]
  |  Branch (72:59): [True: 395k, False: 1.94M]
  ------------------
   73|  4.17M|        unsigned ca, cl;
   74|       |
   75|  4.17M|#define MERGE_CTX(dir, type, no_val) \
   76|  4.17M|        c##dir = *(const type *) dir != no_val; \
   77|  4.17M|        break
   78|       |
   79|  4.17M|        switch (t_dim->lw) {
   80|       |        /* For some reason the MSVC CRT _wassert() function is not flagged as
   81|       |         * __declspec(noreturn), so when using those headers the compiler will
   82|       |         * expect execution to continue after an assertion has been triggered
   83|       |         * and will therefore complain about the use of uninitialized variables
   84|       |         * when compiled in debug mode if we put the default case at the end. */
   85|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (85:9): [True: 0, False: 4.17M]
  |  Branch (85:18): [Folded, False: 0]
  ------------------
   86|  1.01M|        case TX_4X4:   MERGE_CTX(a, uint8_t,  0x40);
  ------------------
  |  |   76|  1.01M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  1.01M|        break
  ------------------
  |  Branch (86:9): [True: 1.01M, False: 3.15M]
  ------------------
   87|   924k|        case TX_8X8:   MERGE_CTX(a, uint16_t, 0x4040);
  ------------------
  |  |   76|   924k|        c##dir = *(const type *) dir != no_val; \
  |  |   77|   924k|        break
  ------------------
  |  Branch (87:9): [True: 924k, False: 3.24M]
  ------------------
   88|   970k|        case TX_16X16: MERGE_CTX(a, uint32_t, 0x40404040U);
  ------------------
  |  |   76|   970k|        c##dir = *(const type *) dir != no_val; \
  |  |   77|   970k|        break
  ------------------
  |  Branch (88:9): [True: 970k, False: 3.20M]
  ------------------
   89|  1.25M|        case TX_32X32: MERGE_CTX(a, uint64_t, 0x4040404040404040ULL);
  ------------------
  |  |   76|  1.25M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  1.25M|        break
  ------------------
  |  Branch (89:9): [True: 1.25M, False: 2.91M]
  ------------------
   90|  4.17M|        }
   91|  4.17M|        switch (t_dim->lh) {
   92|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (92:9): [True: 0, False: 4.17M]
  |  Branch (92:18): [Folded, False: 0]
  ------------------
   93|  1.29M|        case TX_4X4:   MERGE_CTX(l, uint8_t,  0x40);
  ------------------
  |  |   76|  1.29M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  1.29M|        break
  ------------------
  |  Branch (93:9): [True: 1.29M, False: 2.87M]
  ------------------
   94|  1.05M|        case TX_8X8:   MERGE_CTX(l, uint16_t, 0x4040);
  ------------------
  |  |   76|  1.05M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  1.05M|        break
  ------------------
  |  Branch (94:9): [True: 1.05M, False: 3.12M]
  ------------------
   95|   793k|        case TX_16X16: MERGE_CTX(l, uint32_t, 0x40404040U);
  ------------------
  |  |   76|   793k|        c##dir = *(const type *) dir != no_val; \
  |  |   77|   793k|        break
  ------------------
  |  Branch (95:9): [True: 793k, False: 3.37M]
  ------------------
   96|  1.03M|        case TX_32X32: MERGE_CTX(l, uint64_t, 0x4040404040404040ULL);
  ------------------
  |  |   76|  1.03M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  1.03M|        break
  ------------------
  |  Branch (96:9): [True: 1.03M, False: 3.14M]
  ------------------
   97|  4.17M|        }
   98|  4.17M|#undef MERGE_CTX
   99|       |
  100|  4.17M|        return 7 + not_one_blk * 3 + ca + cl;
  101|  4.17M|    } else if (b_dim[2] == t_dim->lw && b_dim[3] == t_dim->lh) {
  ------------------
  |  Branch (101:16): [True: 1.95M, False: 2.19M]
  |  Branch (101:41): [True: 1.83M, False: 118k]
  ------------------
  102|  1.83M|        return 0;
  103|  2.31M|    } else {
  104|  2.31M|        unsigned la, ll;
  105|       |
  106|  2.31M|#define MERGE_CTX(dir, type, tx) \
  107|  2.31M|        if (tx == TX_64X64) { \
  108|  2.31M|            uint64_t tmp = *(const uint64_t *) dir; \
  109|  2.31M|            tmp |= *(const uint64_t *) &dir[8]; \
  110|  2.31M|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  111|  2.31M|        } else \
  112|  2.31M|            l##dir = *(const type *) dir; \
  113|  2.31M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  114|  2.31M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  115|  2.31M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  116|  2.31M|        break
  117|       |
  118|  2.31M|        switch (t_dim->lw) {
  119|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (119:9): [True: 0, False: 2.31M]
  |  Branch (119:18): [Folded, False: 0]
  ------------------
  120|  1.09M|        case TX_4X4:   MERGE_CTX(a, uint8_t,  TX_4X4);
  ------------------
  |  |  107|  1.09M|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 1.09M]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  1.09M|            l##dir = *(const type *) dir; \
  |  |  113|  1.09M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 1.09M]
  |  |  ------------------
  |  |  114|  1.09M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 1.09M]
  |  |  ------------------
  |  |  115|  1.09M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [Folded, False: 1.09M]
  |  |  ------------------
  |  |  116|  1.09M|        break
  ------------------
  |  Branch (120:9): [True: 1.09M, False: 1.21M]
  ------------------
  121|   703k|        case TX_8X8:   MERGE_CTX(a, uint16_t, TX_8X8);
  ------------------
  |  |  107|   703k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 703k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   703k|            l##dir = *(const type *) dir; \
  |  |  113|   703k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 703k]
  |  |  ------------------
  |  |  114|   703k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 703k]
  |  |  ------------------
  |  |  115|   703k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 703k, Folded]
  |  |  ------------------
  |  |  116|   703k|        break
  ------------------
  |  Branch (121:9): [True: 703k, False: 1.61M]
  ------------------
  122|   384k|        case TX_16X16: MERGE_CTX(a, uint32_t, TX_16X16);
  ------------------
  |  |  107|   384k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 384k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   384k|            l##dir = *(const type *) dir; \
  |  |  113|   384k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 384k]
  |  |  ------------------
  |  |  114|   384k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 384k, Folded]
  |  |  ------------------
  |  |  115|   384k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 384k, Folded]
  |  |  ------------------
  |  |  116|   384k|        break
  ------------------
  |  Branch (122:9): [True: 384k, False: 1.93M]
  ------------------
  123|  33.9k|        case TX_32X32: MERGE_CTX(a, uint32_t, TX_32X32);
  ------------------
  |  |  107|  33.9k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 33.9k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  33.9k|            l##dir = *(const type *) dir; \
  |  |  113|  33.9k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [True: 33.9k, Folded]
  |  |  ------------------
  |  |  114|  33.9k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 33.9k, Folded]
  |  |  ------------------
  |  |  115|  33.9k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 33.9k, Folded]
  |  |  ------------------
  |  |  116|  33.9k|        break
  ------------------
  |  Branch (123:9): [True: 33.9k, False: 2.28M]
  ------------------
  124|  94.2k|        case TX_64X64: MERGE_CTX(a, uint32_t, TX_64X64);
  ------------------
  |  |  107|  94.2k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  108|  94.2k|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|  94.2k|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|  94.2k|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|  94.2k|        } else \
  |  |  112|  94.2k|            l##dir = *(const type *) dir; \
  |  |  113|  94.2k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 94.2k]
  |  |  ------------------
  |  |  114|  94.2k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  115|  94.2k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  116|  94.2k|        break
  ------------------
  |  Branch (124:9): [True: 94.2k, False: 2.22M]
  ------------------
  125|  2.31M|        }
  126|  2.31M|        switch (t_dim->lh) {
  127|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (127:9): [True: 0, False: 2.31M]
  |  Branch (127:18): [Folded, False: 0]
  ------------------
  128|  1.10M|        case TX_4X4:   MERGE_CTX(l, uint8_t,  TX_4X4);
  ------------------
  |  |  107|  1.10M|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 1.10M]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  1.10M|            l##dir = *(const type *) dir; \
  |  |  113|  1.10M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 1.10M]
  |  |  ------------------
  |  |  114|  1.10M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 1.10M]
  |  |  ------------------
  |  |  115|  1.10M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [Folded, False: 1.10M]
  |  |  ------------------
  |  |  116|  1.10M|        break
  ------------------
  |  Branch (128:9): [True: 1.10M, False: 1.20M]
  ------------------
  129|   701k|        case TX_8X8:   MERGE_CTX(l, uint16_t, TX_8X8);
  ------------------
  |  |  107|   701k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 701k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   701k|            l##dir = *(const type *) dir; \
  |  |  113|   701k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 701k]
  |  |  ------------------
  |  |  114|   701k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 701k]
  |  |  ------------------
  |  |  115|   701k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 701k, Folded]
  |  |  ------------------
  |  |  116|   701k|        break
  ------------------
  |  Branch (129:9): [True: 701k, False: 1.61M]
  ------------------
  130|   376k|        case TX_16X16: MERGE_CTX(l, uint32_t, TX_16X16);
  ------------------
  |  |  107|   376k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 376k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   376k|            l##dir = *(const type *) dir; \
  |  |  113|   376k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 376k]
  |  |  ------------------
  |  |  114|   376k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 376k, Folded]
  |  |  ------------------
  |  |  115|   376k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 376k, Folded]
  |  |  ------------------
  |  |  116|   376k|        break
  ------------------
  |  Branch (130:9): [True: 376k, False: 1.93M]
  ------------------
  131|  33.5k|        case TX_32X32: MERGE_CTX(l, uint32_t, TX_32X32);
  ------------------
  |  |  107|  33.5k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 33.5k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  33.5k|            l##dir = *(const type *) dir; \
  |  |  113|  33.5k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [True: 33.5k, Folded]
  |  |  ------------------
  |  |  114|  33.5k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 33.5k, Folded]
  |  |  ------------------
  |  |  115|  33.5k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 33.5k, Folded]
  |  |  ------------------
  |  |  116|  33.5k|        break
  ------------------
  |  Branch (131:9): [True: 33.5k, False: 2.28M]
  ------------------
  132|  94.2k|        case TX_64X64: MERGE_CTX(l, uint32_t, TX_64X64);
  ------------------
  |  |  107|  94.2k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  108|  94.2k|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|  94.2k|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|  94.2k|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|  94.2k|        } else \
  |  |  112|  94.2k|            l##dir = *(const type *) dir; \
  |  |  113|  94.2k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 94.2k]
  |  |  ------------------
  |  |  114|  94.2k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  115|  94.2k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 94.2k, Folded]
  |  |  ------------------
  |  |  116|  94.2k|        break
  ------------------
  |  Branch (132:9): [True: 94.2k, False: 2.22M]
  ------------------
  133|  2.31M|        }
  134|  2.31M|#undef MERGE_CTX
  135|       |
  136|  2.31M|        return dav1d_skip_ctx[umin(la & 0x3F, 4)][umin(ll & 0x3F, 4)];
  137|  2.31M|    }
  138|  8.32M|}
recon_tmpl.c:get_lo_ctx:
  304|  80.8M|{
  305|  80.8M|    unsigned mag = levels[0 * stride + 1] + levels[1 * stride + 0];
  306|  80.8M|    unsigned offset;
  307|  80.8M|    if (tx_class == TX_CLASS_2D) {
  ------------------
  |  Branch (307:9): [True: 76.0M, False: 4.76M]
  ------------------
  308|  76.0M|        mag += levels[1 * stride + 1];
  309|  76.0M|        *hi_mag = mag;
  310|  76.0M|        mag += levels[0 * stride + 2] + levels[2 * stride + 0];
  311|  76.0M|        offset = ctx_offsets[umin(y, 4)][umin(x, 4)];
  312|  76.0M|    } else {
  313|  4.76M|        mag += levels[0 * stride + 2];
  314|  4.76M|        *hi_mag = mag;
  315|  4.76M|        mag += levels[0 * stride + 3] + levels[0 * stride + 4];
  316|  4.76M|        offset = 26 + (y > 1 ? 10 : y * 5);
  ------------------
  |  Branch (316:24): [True: 2.15M, False: 2.60M]
  ------------------
  317|  4.76M|    }
  318|  80.8M|    return offset + (mag > 512 ? 4 : (mag + 64) >> 7);
  ------------------
  |  Branch (318:22): [True: 5.71M, False: 75.0M]
  ------------------
  319|  80.8M|}
recon_tmpl.c:get_dc_sign_ctx:
  143|  3.26M|{
  144|  3.26M|    uint64_t mask = 0xC0C0C0C0C0C0C0C0ULL, mul = 0x0101010101010101ULL;
  145|  3.26M|    int s;
  146|       |
  147|  3.26M|#if ARCH_X86_64 && defined(__GNUC__)
  148|       |    /* Coerce compilers into producing better code. For some reason
  149|       |     * every x86-64 compiler is awful at handling 64-bit constants. */
  150|  3.26M|    __asm__("" : "+r"(mask), "+r"(mul));
  151|  3.26M|#endif
  152|       |
  153|  3.26M|    switch(tx) {
  154|      0|    default: assert(0); /* fall-through */
  ------------------
  |  Branch (154:5): [True: 0, False: 3.26M]
  |  Branch (154:14): [Folded, False: 0]
  ------------------
  155|   895k|    case TX_4X4: {
  ------------------
  |  Branch (155:5): [True: 895k, False: 2.36M]
  ------------------
  156|   895k|        int t = *(const uint8_t *) a >> 6;
  157|   895k|        t    += *(const uint8_t *) l >> 6;
  158|   895k|        s = t - 1 - 1;
  159|   895k|        break;
  160|      0|    }
  161|   461k|    case TX_8X8: {
  ------------------
  |  Branch (161:5): [True: 461k, False: 2.79M]
  ------------------
  162|   461k|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  163|   461k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  164|   461k|        t *= 0x04040404U;
  165|   461k|        s = (int) (t >> 24) - 2 - 2;
  166|   461k|        break;
  167|      0|    }
  168|   313k|    case TX_16X16: {
  ------------------
  |  Branch (168:5): [True: 313k, False: 2.94M]
  ------------------
  169|   313k|        uint32_t t = (*(const uint32_t *) a & (uint32_t) mask) >> 6;
  170|   313k|        t         += (*(const uint32_t *) l & (uint32_t) mask) >> 6;
  171|   313k|        t *= (uint32_t) mul;
  172|   313k|        s = (int) (t >> 24) - 4 - 4;
  173|   313k|        break;
  174|      0|    }
  175|   282k|    case TX_32X32: {
  ------------------
  |  Branch (175:5): [True: 282k, False: 2.97M]
  ------------------
  176|   282k|        uint64_t t = (*(const uint64_t *) a & mask) >> 6;
  177|   282k|        t         += (*(const uint64_t *) l & mask) >> 6;
  178|   282k|        t *= mul;
  179|   282k|        s = (int) (t >> 56) - 8 - 8;
  180|   282k|        break;
  181|      0|    }
  182|  94.0k|    case TX_64X64: {
  ------------------
  |  Branch (182:5): [True: 94.0k, False: 3.16M]
  ------------------
  183|  94.0k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  184|  94.0k|        t         += (*(const uint64_t *) &a[8] & mask) >> 6;
  185|  94.0k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  186|  94.0k|        t         += (*(const uint64_t *) &l[8] & mask) >> 6;
  187|  94.0k|        t *= mul;
  188|  94.0k|        s = (int) (t >> 56) - 16 - 16;
  189|  94.0k|        break;
  190|      0|    }
  191|   103k|    case RTX_4X8: {
  ------------------
  |  Branch (191:5): [True: 103k, False: 3.15M]
  ------------------
  192|   103k|        uint32_t t = *(const uint8_t  *) a & (uint32_t) mask;
  193|   103k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  194|   103k|        t *= 0x04040404U;
  195|   103k|        s = (int) (t >> 24) - 1 - 2;
  196|   103k|        break;
  197|      0|    }
  198|   169k|    case RTX_8X4: {
  ------------------
  |  Branch (198:5): [True: 169k, False: 3.09M]
  ------------------
  199|   169k|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  200|   169k|        t         += *(const uint8_t  *) l & (uint32_t) mask;
  201|   169k|        t *= 0x04040404U;
  202|   169k|        s = (int) (t >> 24) - 2 - 1;
  203|   169k|        break;
  204|      0|    }
  205|   114k|    case RTX_8X16: {
  ------------------
  |  Branch (205:5): [True: 114k, False: 3.14M]
  ------------------
  206|   114k|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  207|   114k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  208|   114k|        t = (t >> 6) * (uint32_t) mul;
  209|   114k|        s = (int) (t >> 24) - 2 - 4;
  210|   114k|        break;
  211|      0|    }
  212|   225k|    case RTX_16X8: {
  ------------------
  |  Branch (212:5): [True: 225k, False: 3.03M]
  ------------------
  213|   225k|        uint32_t t = *(const uint32_t *) a & (uint32_t) mask;
  214|   225k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  215|   225k|        t = (t >> 6) * (uint32_t) mul;
  216|   225k|        s = (int) (t >> 24) - 4 - 2;
  217|   225k|        break;
  218|      0|    }
  219|  54.2k|    case RTX_16X32: {
  ------------------
  |  Branch (219:5): [True: 54.2k, False: 3.20M]
  ------------------
  220|  54.2k|        uint64_t t = *(const uint32_t *) a & (uint32_t) mask;
  221|  54.2k|        t         += *(const uint64_t *) l & mask;
  222|  54.2k|        t = (t >> 6) * mul;
  223|  54.2k|        s = (int) (t >> 56) - 4 - 8;
  224|  54.2k|        break;
  225|      0|    }
  226|   112k|    case RTX_32X16: {
  ------------------
  |  Branch (226:5): [True: 112k, False: 3.14M]
  ------------------
  227|   112k|        uint64_t t = *(const uint64_t *) a & mask;
  228|   112k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  229|   112k|        t = (t >> 6) * mul;
  230|   112k|        s = (int) (t >> 56) - 8 - 4;
  231|   112k|        break;
  232|      0|    }
  233|  7.83k|    case RTX_32X64: {
  ------------------
  |  Branch (233:5): [True: 7.83k, False: 3.25M]
  ------------------
  234|  7.83k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  235|  7.83k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  236|  7.83k|        t         += (*(const uint64_t *) &l[8] & mask) >> 6;
  237|  7.83k|        t *= mul;
  238|  7.83k|        s = (int) (t >> 56) - 8 - 16;
  239|  7.83k|        break;
  240|      0|    }
  241|  34.4k|    case RTX_64X32: {
  ------------------
  |  Branch (241:5): [True: 34.4k, False: 3.22M]
  ------------------
  242|  34.4k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  243|  34.4k|        t         += (*(const uint64_t *) &a[8] & mask) >> 6;
  244|  34.4k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  245|  34.4k|        t *= mul;
  246|  34.4k|        s = (int) (t >> 56) - 16 - 8;
  247|  34.4k|        break;
  248|      0|    }
  249|  77.5k|    case RTX_4X16: {
  ------------------
  |  Branch (249:5): [True: 77.5k, False: 3.18M]
  ------------------
  250|  77.5k|        uint32_t t = *(const uint8_t  *) a & (uint32_t) mask;
  251|  77.5k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  252|  77.5k|        t = (t >> 6) * (uint32_t) mul;
  253|  77.5k|        s = (int) (t >> 24) - 1 - 4;
  254|  77.5k|        break;
  255|      0|    }
  256|   156k|    case RTX_16X4: {
  ------------------
  |  Branch (256:5): [True: 156k, False: 3.10M]
  ------------------
  257|   156k|        uint32_t t = *(const uint32_t *) a & (uint32_t) mask;
  258|   156k|        t         += *(const uint8_t  *) l & (uint32_t) mask;
  259|   156k|        t = (t >> 6) * (uint32_t) mul;
  260|   156k|        s = (int) (t >> 24) - 4 - 1;
  261|   156k|        break;
  262|      0|    }
  263|  37.4k|    case RTX_8X32: {
  ------------------
  |  Branch (263:5): [True: 37.4k, False: 3.22M]
  ------------------
  264|  37.4k|        uint64_t t = *(const uint16_t *) a & (uint32_t) mask;
  265|  37.4k|        t         += *(const uint64_t *) l & mask;
  266|  37.4k|        t = (t >> 6) * mul;
  267|  37.4k|        s = (int) (t >> 56) - 2 - 8;
  268|  37.4k|        break;
  269|      0|    }
  270|  93.3k|    case RTX_32X8: {
  ------------------
  |  Branch (270:5): [True: 93.3k, False: 3.16M]
  ------------------
  271|  93.3k|        uint64_t t = *(const uint64_t *) a & mask;
  272|  93.3k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  273|  93.3k|        t = (t >> 6) * mul;
  274|  93.3k|        s = (int) (t >> 56) - 8 - 2;
  275|  93.3k|        break;
  276|      0|    }
  277|  10.2k|    case RTX_16X64: {
  ------------------
  |  Branch (277:5): [True: 10.2k, False: 3.25M]
  ------------------
  278|  10.2k|        uint64_t t = *(const uint32_t *) a & (uint32_t) mask;
  279|  10.2k|        t         += *(const uint64_t *) &l[0] & mask;
  280|  10.2k|        t = (t >> 6) + ((*(const uint64_t *) &l[8] & mask) >> 6);
  281|  10.2k|        t *= mul;
  282|  10.2k|        s = (int) (t >> 56) - 4 - 16;
  283|  10.2k|        break;
  284|      0|    }
  285|  17.9k|    case RTX_64X16: {
  ------------------
  |  Branch (285:5): [True: 17.9k, False: 3.24M]
  ------------------
  286|  17.9k|        uint64_t t = *(const uint64_t *) &a[0] & mask;
  287|  17.9k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  288|  17.9k|        t = (t >> 6) + ((*(const uint64_t *) &a[8] & mask) >> 6);
  289|  17.9k|        t *= mul;
  290|  17.9k|        s = (int) (t >> 56) - 16 - 4;
  291|  17.9k|        break;
  292|      0|    }
  293|  3.26M|    }
  294|       |
  295|  3.26M|    return (s != 0) + (s > 0);
  296|  3.26M|}
recon_tmpl.c:read_golomb:
   49|  1.46M|static inline unsigned read_golomb(MsacContext *const msac) {
   50|  1.46M|    int len = 0;
   51|  1.46M|    unsigned val = 1;
   52|       |
   53|  3.22M|    while (!dav1d_msac_decode_bool_equi(msac) && len < 32) len++;
  ------------------
  |  |   53|  3.22M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (53:12): [True: 1.76M, False: 1.46M]
  |  Branch (53:50): [True: 1.76M, False: 70]
  ------------------
   54|  3.22M|    while (len--) val = (val << 1) + dav1d_msac_decode_bool_equi(msac);
  ------------------
  |  |   53|  1.76M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (54:12): [True: 1.76M, False: 1.46M]
  ------------------
   55|       |
   56|  1.46M|    return val - 1;
   57|  1.46M|}
recon_tmpl.c:mc:
  944|  4.19M|{
  945|  4.19M|    assert((dst8 != NULL) ^ (dst16 != NULL));
  ------------------
  |  Branch (945:5): [True: 4.19M, False: 0]
  ------------------
  946|  4.19M|    const Dav1dFrameContext *const f = t->f;
  947|  4.19M|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (947:24): [True: 2.03M, False: 2.16M]
  |  Branch (947:32): [True: 525k, False: 1.50M]
  ------------------
  948|  4.19M|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (948:24): [True: 2.03M, False: 2.16M]
  |  Branch (948:32): [True: 530k, False: 1.50M]
  ------------------
  949|  4.19M|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
  950|  4.19M|    const int mvx = mv.x, mvy = mv.y;
  951|  4.19M|    const int mx = mvx & (15 >> !ss_hor), my = mvy & (15 >> !ss_ver);
  952|  4.19M|    ptrdiff_t ref_stride = refp->p.stride[!!pl];
  953|  4.19M|    const pixel *ref;
  954|       |
  955|  4.19M|    if (refp->p.p.w == f->cur.p.w && refp->p.p.h == f->cur.p.h) {
  ------------------
  |  Branch (955:9): [True: 2.98M, False: 1.21M]
  |  Branch (955:38): [True: 2.91M, False: 71.9k]
  ------------------
  956|  2.91M|        const int dx = bx * h_mul + (mvx >> (3 + ss_hor));
  957|  2.91M|        const int dy = by * v_mul + (mvy >> (3 + ss_ver));
  958|  2.91M|        int w, h;
  959|       |
  960|  2.91M|        if (refp->p.data[0] != f->cur.data[0]) { // i.e. not for intrabc
  ------------------
  |  Branch (960:13): [True: 1.70M, False: 1.20M]
  ------------------
  961|  1.70M|            w = (f->cur.p.w + ss_hor) >> ss_hor;
  962|  1.70M|            h = (f->cur.p.h + ss_ver) >> ss_ver;
  963|  1.70M|        } else {
  964|  1.20M|            w = f->bw * 4 >> ss_hor;
  965|  1.20M|            h = f->bh * 4 >> ss_ver;
  966|  1.20M|        }
  967|  2.91M|        if (dx < !!mx * 3 || dy < !!my * 3 ||
  ------------------
  |  Branch (967:13): [True: 41.4k, False: 2.86M]
  |  Branch (967:30): [True: 67.4k, False: 2.80M]
  ------------------
  968|  2.80M|            dx + bw4 * h_mul + !!mx * 4 > w ||
  ------------------
  |  Branch (968:13): [True: 98.5k, False: 2.70M]
  ------------------
  969|  2.70M|            dy + bh4 * v_mul + !!my * 4 > h)
  ------------------
  |  Branch (969:13): [True: 262k, False: 2.44M]
  ------------------
  970|   469k|        {
  971|   469k|            pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|   469k|#define bitfn(x) x##_8bpc
  ------------------
  972|   469k|            f->dsp->mc.emu_edge(bw4 * h_mul + !!mx * 7, bh4 * v_mul + !!my * 7,
  973|   469k|                                w, h, dx - !!mx * 3, dy - !!my * 3,
  974|   469k|                                emu_edge_buf, 192 * sizeof(pixel),
  975|   469k|                                refp->p.data[pl], ref_stride);
  976|   469k|            ref = &emu_edge_buf[192 * !!my * 3 + !!mx * 3];
  977|   469k|            ref_stride = 192 * sizeof(pixel);
  978|  2.44M|        } else {
  979|  2.44M|            ref = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * dy + dx;
  ------------------
  |  |   53|  2.44M|#define PXSTRIDE(x) (x)
  ------------------
  980|  2.44M|        }
  981|       |
  982|  2.91M|        if (dst8 != NULL) {
  ------------------
  |  Branch (982:13): [True: 2.48M, False: 423k]
  ------------------
  983|  2.48M|            f->dsp->mc.mc[filter_2d](dst8, dst_stride, ref, ref_stride, bw4 * h_mul,
  984|  2.48M|                                     bh4 * v_mul, mx << !ss_hor, my << !ss_ver
  985|  2.48M|                                     HIGHBD_CALL_SUFFIX);
  986|  2.48M|        } else {
  987|   423k|            f->dsp->mc.mct[filter_2d](dst16, ref, ref_stride, bw4 * h_mul,
  988|   423k|                                      bh4 * v_mul, mx << !ss_hor, my << !ss_ver
  989|   423k|                                      HIGHBD_CALL_SUFFIX);
  990|   423k|        }
  991|  2.91M|    } else {
  992|  1.28M|        assert(refp != &f->sr_cur);
  ------------------
  |  Branch (992:9): [True: 1.28M, False: 0]
  ------------------
  993|       |
  994|  1.28M|        const int orig_pos_y = (by * v_mul << 4) + mvy * (1 << !ss_ver);
  995|  1.28M|        const int orig_pos_x = (bx * h_mul << 4) + mvx * (1 << !ss_hor);
  996|  1.28M|#define scale_mv(res, val, scale) do { \
  997|  1.28M|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  998|  1.28M|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  999|  1.28M|        } while (0)
 1000|  1.28M|        int pos_y, pos_x;
 1001|  1.28M|        scale_mv(pos_x, orig_pos_x, f->svc[refidx][0].scale);
  ------------------
  |  |  996|  1.28M|#define scale_mv(res, val, scale) do { \
  |  |  997|  1.28M|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  |  |  998|  1.28M|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  |  |  999|  1.28M|        } while (0)
  |  |  ------------------
  |  |  |  Branch (999:18): [Folded, False: 1.28M]
  |  |  ------------------
  ------------------
 1002|  1.28M|        scale_mv(pos_y, orig_pos_y, f->svc[refidx][1].scale);
  ------------------
  |  |  996|  1.28M|#define scale_mv(res, val, scale) do { \
  |  |  997|  1.28M|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  |  |  998|  1.28M|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  |  |  999|  1.28M|        } while (0)
  |  |  ------------------
  |  |  |  Branch (999:18): [Folded, False: 1.28M]
  |  |  ------------------
  ------------------
 1003|  1.28M|#undef scale_mv
 1004|  1.28M|        const int left = pos_x >> 10;
 1005|  1.28M|        const int top = pos_y >> 10;
 1006|  1.28M|        const int right =
 1007|  1.28M|            ((pos_x + (bw4 * h_mul - 1) * f->svc[refidx][0].step) >> 10) + 1;
 1008|  1.28M|        const int bottom =
 1009|  1.28M|            ((pos_y + (bh4 * v_mul - 1) * f->svc[refidx][1].step) >> 10) + 1;
 1010|       |
 1011|  1.28M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.28M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.28M]
  |  |  ------------------
  |  |   35|  1.28M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.28M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1012|      0|            printf("Off %dx%d [%d,%d,%d], size %dx%d [%d,%d]\n",
 1013|      0|                   left, top, orig_pos_x, f->svc[refidx][0].scale, refidx,
 1014|      0|                   right-left, bottom-top,
 1015|      0|                   f->svc[refidx][0].step, f->svc[refidx][1].step);
 1016|       |
 1017|  1.28M|        const int w = (refp->p.p.w + ss_hor) >> ss_hor;
 1018|  1.28M|        const int h = (refp->p.p.h + ss_ver) >> ss_ver;
 1019|  1.28M|        if (left < 3 || top < 3 || right + 4 > w || bottom + 4 > h) {
  ------------------
  |  Branch (1019:13): [True: 138k, False: 1.14M]
  |  Branch (1019:25): [True: 290k, False: 853k]
  |  Branch (1019:36): [True: 98.5k, False: 754k]
  |  Branch (1019:53): [True: 86.9k, False: 668k]
  ------------------
 1020|   614k|            pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|   614k|#define bitfn(x) x##_8bpc
  ------------------
 1021|   614k|            f->dsp->mc.emu_edge(right - left + 7, bottom - top + 7,
 1022|   614k|                                w, h, left - 3, top - 3,
 1023|   614k|                                emu_edge_buf, 320 * sizeof(pixel),
 1024|   614k|                                refp->p.data[pl], ref_stride);
 1025|   614k|            ref = &emu_edge_buf[320 * 3 + 3];
 1026|   614k|            ref_stride = 320 * sizeof(pixel);
 1027|   614k|            if (DEBUG_BLOCK_INFO) printf("Emu\n");
  ------------------
  |  |   34|   614k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 614k]
  |  |  ------------------
  |  |   35|   614k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   614k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1028|   668k|        } else {
 1029|   668k|            ref = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * top + left;
  ------------------
  |  |   53|   668k|#define PXSTRIDE(x) (x)
  ------------------
 1030|   668k|        }
 1031|       |
 1032|  1.28M|        if (dst8 != NULL) {
  ------------------
  |  Branch (1032:13): [True: 935k, False: 346k]
  ------------------
 1033|   935k|            f->dsp->mc.mc_scaled[filter_2d](dst8, dst_stride, ref, ref_stride,
 1034|   935k|                                            bw4 * h_mul, bh4 * v_mul,
 1035|   935k|                                            pos_x & 0x3ff, pos_y & 0x3ff,
 1036|   935k|                                            f->svc[refidx][0].step,
 1037|   935k|                                            f->svc[refidx][1].step
 1038|   935k|                                            HIGHBD_CALL_SUFFIX);
 1039|   935k|        } else {
 1040|   346k|            f->dsp->mc.mct_scaled[filter_2d](dst16, ref, ref_stride,
 1041|   346k|                                             bw4 * h_mul, bh4 * v_mul,
 1042|   346k|                                             pos_x & 0x3ff, pos_y & 0x3ff,
 1043|   346k|                                             f->svc[refidx][0].step,
 1044|   346k|                                             f->svc[refidx][1].step
 1045|   346k|                                             HIGHBD_CALL_SUFFIX);
 1046|   346k|        }
 1047|  1.28M|    }
 1048|       |
 1049|  4.19M|    return 0;
 1050|  4.19M|}
recon_tmpl.c:warp_affine:
 1120|   142k|{
 1121|   142k|    assert((dst8 != NULL) ^ (dst16 != NULL));
  ------------------
  |  Branch (1121:5): [True: 142k, False: 0]
  ------------------
 1122|   142k|    const Dav1dFrameContext *const f = t->f;
 1123|   142k|    const Dav1dDSPContext *const dsp = f->dsp;
 1124|   142k|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (1124:24): [True: 48.4k, False: 94.3k]
  |  Branch (1124:32): [True: 17.4k, False: 30.9k]
  ------------------
 1125|   142k|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (1125:24): [True: 48.4k, False: 94.3k]
  |  Branch (1125:32): [True: 17.5k, False: 30.9k]
  ------------------
 1126|   142k|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
 1127|   142k|    assert(!((b_dim[0] * h_mul) & 7) && !((b_dim[1] * v_mul) & 7));
  ------------------
  |  Branch (1127:5): [True: 142k, False: 0]
  |  Branch (1127:5): [True: 142k, False: 0]
  ------------------
 1128|   142k|    const int32_t *const mat = wmp->matrix;
 1129|   142k|    const int width = (refp->p.p.w + ss_hor) >> ss_hor;
 1130|   142k|    const int height = (refp->p.p.h + ss_ver) >> ss_ver;
 1131|       |
 1132|   987k|    for (int y = 0; y < b_dim[1] * v_mul; y += 8) {
  ------------------
  |  Branch (1132:21): [True: 845k, False: 142k]
  ------------------
 1133|   845k|        const int src_y = t->by * 4 + ((y + 4) << ss_ver);
 1134|   845k|        const int64_t mat3_y = (int64_t) mat[3] * src_y + mat[0];
 1135|   845k|        const int64_t mat5_y = (int64_t) mat[5] * src_y + mat[1];
 1136|  7.09M|        for (int x = 0; x < b_dim[0] * h_mul; x += 8) {
  ------------------
  |  Branch (1136:25): [True: 6.24M, False: 845k]
  ------------------
 1137|       |            // calculate transformation relative to center of 8x8 block in
 1138|       |            // luma pixel units
 1139|  6.24M|            const int src_x = t->bx * 4 + ((x + 4) << ss_hor);
 1140|  6.24M|            const int64_t mvx = ((int64_t) mat[2] * src_x + mat3_y) >> ss_hor;
 1141|  6.24M|            const int64_t mvy = ((int64_t) mat[4] * src_x + mat5_y) >> ss_ver;
 1142|       |
 1143|  6.24M|            const int dx = (int) (mvx >> 16) - 4;
 1144|  6.24M|            const int mx = (((int) mvx & 0xffff) - wmp->u.p.alpha * 4 -
 1145|  6.24M|                                                   wmp->u.p.beta  * 7) & ~0x3f;
 1146|  6.24M|            const int dy = (int) (mvy >> 16) - 4;
 1147|  6.24M|            const int my = (((int) mvy & 0xffff) - wmp->u.p.gamma * 4 -
 1148|  6.24M|                                                   wmp->u.p.delta * 4) & ~0x3f;
 1149|       |
 1150|  6.24M|            const pixel *ref_ptr;
 1151|  6.24M|            ptrdiff_t ref_stride = refp->p.stride[!!pl];
 1152|       |
 1153|  6.24M|            if (dx < 3 || dx + 8 + 4 > width || dy < 3 || dy + 8 + 4 > height) {
  ------------------
  |  Branch (1153:17): [True: 630k, False: 5.61M]
  |  Branch (1153:27): [True: 1.76M, False: 3.85M]
  |  Branch (1153:49): [True: 75.1k, False: 3.77M]
  |  Branch (1153:59): [True: 204k, False: 3.57M]
  ------------------
 1154|  2.67M|                pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|  2.67M|#define bitfn(x) x##_8bpc
  ------------------
 1155|  2.67M|                f->dsp->mc.emu_edge(15, 15, width, height, dx - 3, dy - 3,
 1156|  2.67M|                                    emu_edge_buf, 32 * sizeof(pixel),
 1157|  2.67M|                                    refp->p.data[pl], ref_stride);
 1158|  2.67M|                ref_ptr = &emu_edge_buf[32 * 3 + 3];
 1159|  2.67M|                ref_stride = 32 * sizeof(pixel);
 1160|  3.57M|            } else {
 1161|  3.57M|                ref_ptr = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * dy + dx;
  ------------------
  |  |   53|  3.57M|#define PXSTRIDE(x) (x)
  ------------------
 1162|  3.57M|            }
 1163|  6.24M|            if (dst16 != NULL)
  ------------------
  |  Branch (1163:17): [True: 71.2k, False: 6.17M]
  ------------------
 1164|  71.2k|                dsp->mc.warp8x8t(&dst16[x], dstride, ref_ptr, ref_stride,
 1165|  71.2k|                                 wmp->u.abcd, mx, my HIGHBD_CALL_SUFFIX);
 1166|  6.17M|            else
 1167|  6.17M|                dsp->mc.warp8x8(&dst8[x], dstride, ref_ptr, ref_stride,
 1168|  6.17M|                                wmp->u.abcd, mx, my HIGHBD_CALL_SUFFIX);
 1169|  6.24M|        }
 1170|   845k|        if (dst8) dst8  += 8 * PXSTRIDE(dstride);
  ------------------
  |  |   53|   830k|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (1170:13): [True: 830k, False: 14.5k]
  ------------------
 1171|  14.5k|        else      dst16 += 8 * dstride;
 1172|   845k|    }
 1173|   142k|    return 0;
 1174|   142k|}
recon_tmpl.c:obmc:
 1056|   334k|{
 1057|   334k|    assert(!(t->bx & 1) && !(t->by & 1));
  ------------------
  |  Branch (1057:5): [True: 334k, False: 0]
  |  Branch (1057:5): [True: 334k, False: 0]
  ------------------
 1058|   334k|    const Dav1dFrameContext *const f = t->f;
 1059|   334k|    /*const*/ refmvs_block **r = &t->rt.r[(t->by & 31) + 5];
 1060|   334k|    pixel *const lap = bitfn(t->scratch.lap);
  ------------------
  |  |   51|   334k|#define bitfn(x) x##_8bpc
  ------------------
 1061|   334k|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (1061:24): [True: 159k, False: 175k]
  |  Branch (1061:32): [True: 48.0k, False: 111k]
  ------------------
 1062|   334k|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (1062:24): [True: 159k, False: 175k]
  |  Branch (1062:32): [True: 48.0k, False: 111k]
  ------------------
 1063|   334k|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
 1064|   334k|    int res;
 1065|       |
 1066|   334k|    if (t->by > t->ts->tiling.row_start &&
  ------------------
  |  Branch (1066:9): [True: 302k, False: 32.2k]
  ------------------
 1067|   302k|        (!pl || b_dim[0] * h_mul + b_dim[1] * v_mul >= 16))
  ------------------
  |  Branch (1067:10): [True: 160k, False: 141k]
  |  Branch (1067:17): [True: 112k, False: 29.7k]
  ------------------
 1068|   272k|    {
 1069|   576k|        for (int i = 0, x = 0; x < w4 && i < imin(b_dim[2], 4); ) {
  ------------------
  |  Branch (1069:32): [True: 304k, False: 271k]
  |  Branch (1069:42): [True: 303k, False: 975]
  ------------------
 1070|       |            // only odd blocks are considered for overlap handling, hence +1
 1071|   303k|            const refmvs_block *const a_r = &r[-1][t->bx + x + 1];
 1072|   303k|            const uint8_t *const a_b_dim = dav1d_block_dimensions[a_r->bs];
 1073|   303k|            const int step4 = iclip(a_b_dim[0], 2, 16);
 1074|       |
 1075|   303k|            if (a_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (1075:17): [True: 296k, False: 7.23k]
  ------------------
 1076|   296k|                const int ow4 = imin(step4, b_dim[0]);
 1077|   296k|                const int oh4 = imin(b_dim[1], 16) >> 1;
 1078|   296k|                res = mc(t, lap, NULL, ow4 * h_mul * sizeof(pixel), ow4, (oh4 * 3 + 3) >> 2,
 1079|   296k|                         t->bx + x, t->by, pl, a_r->mv.mv[0],
 1080|   296k|                         &f->refp[a_r->ref.ref[0] - 1], a_r->ref.ref[0] - 1,
 1081|   296k|                         dav1d_filter_2d[t->a->filter[1][bx4 + x + 1]][t->a->filter[0][bx4 + x + 1]]);
 1082|   296k|                if (res) return res;
  ------------------
  |  Branch (1082:21): [True: 0, False: 296k]
  ------------------
 1083|   296k|                f->dsp->mc.blend_h(&dst[x * h_mul], dst_stride, lap,
 1084|   296k|                                   h_mul * ow4, v_mul * oh4);
 1085|   296k|                i++;
 1086|   296k|            }
 1087|   303k|            x += step4;
 1088|   303k|        }
 1089|   272k|    }
 1090|       |
 1091|   334k|    if (t->bx > t->ts->tiling.col_start)
  ------------------
  |  Branch (1091:9): [True: 322k, False: 12.2k]
  ------------------
 1092|   676k|        for (int i = 0, y = 0; y < h4 && i < imin(b_dim[3], 4); ) {
  ------------------
  |  Branch (1092:32): [True: 356k, False: 320k]
  |  Branch (1092:42): [True: 354k, False: 1.82k]
  ------------------
 1093|       |            // only odd blocks are considered for overlap handling, hence +1
 1094|   354k|            const refmvs_block *const l_r = &r[y + 1][t->bx - 1];
 1095|   354k|            const uint8_t *const l_b_dim = dav1d_block_dimensions[l_r->bs];
 1096|   354k|            const int step4 = iclip(l_b_dim[1], 2, 16);
 1097|       |
 1098|   354k|            if (l_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (1098:17): [True: 344k, False: 9.85k]
  ------------------
 1099|   344k|                const int ow4 = imin(b_dim[0], 16) >> 1;
 1100|   344k|                const int oh4 = imin(step4, b_dim[1]);
 1101|   344k|                res = mc(t, lap, NULL, h_mul * ow4 * sizeof(pixel), ow4, oh4,
 1102|   344k|                         t->bx, t->by + y, pl, l_r->mv.mv[0],
 1103|   344k|                         &f->refp[l_r->ref.ref[0] - 1], l_r->ref.ref[0] - 1,
 1104|   344k|                         dav1d_filter_2d[t->l.filter[1][by4 + y + 1]][t->l.filter[0][by4 + y + 1]]);
 1105|   344k|                if (res) return res;
  ------------------
  |  Branch (1105:21): [True: 0, False: 344k]
  ------------------
 1106|   344k|                f->dsp->mc.blend_v(&dst[y * v_mul * PXSTRIDE(dst_stride)],
  ------------------
  |  |   53|   344k|#define PXSTRIDE(x) (x)
  ------------------
 1107|   344k|                                   dst_stride, lap, h_mul * ow4, v_mul * oh4);
 1108|   344k|                i++;
 1109|   344k|            }
 1110|   354k|            y += step4;
 1111|   354k|        }
 1112|   334k|    return 0;
 1113|   334k|}
dav1d_recon_b_intra_16bpc:
 1179|  1.49M|{
 1180|  1.49M|    Dav1dTileState *const ts = t->ts;
 1181|  1.49M|    const Dav1dFrameContext *const f = t->f;
 1182|  1.49M|    const Dav1dDSPContext *const dsp = f->dsp;
 1183|  1.49M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1184|  1.49M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1185|  1.49M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1186|  1.49M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1187|  1.49M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1188|  1.49M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1189|  1.49M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1190|  1.49M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1191|  1.49M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1191:28): [True: 999k, False: 491k]
  ------------------
 1192|   999k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1192:29): [True: 966k, False: 32.7k]
  |  Branch (1192:45): [True: 16.4k, False: 16.3k]
  ------------------
 1193|   983k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1193:29): [True: 961k, False: 21.7k]
  |  Branch (1193:45): [True: 11.3k, False: 10.4k]
  ------------------
 1194|  1.49M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|  1.49M|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
 1196|       |
 1197|       |    // coefficient coding
 1198|  1.49M|    pixel *const edge = bitfn(t->scratch.edge) + 128;
  ------------------
  |  |   77|  1.49M|#define bitfn(x) x##_16bpc
  ------------------
 1199|  1.49M|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
 1200|       |
 1201|  1.49M|    const int intra_edge_filter_flag = f->seq_hdr->intra_edge_filter << 10;
 1202|       |
 1203|  3.04M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (1203:26): [True: 1.55M, False: 1.49M]
  ------------------
 1204|  1.55M|        const int sub_h4 = imin(h4, 16 + init_y);
 1205|  1.55M|        const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
 1206|  3.25M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (1206:30): [True: 1.69M, False: 1.55M]
  ------------------
 1207|  1.69M|            if (b->pal_sz[0]) {
  ------------------
  |  Branch (1207:17): [True: 58.5k, False: 1.63M]
  ------------------
 1208|  58.5k|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1209|  58.5k|                             4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
 1210|  58.5k|                const uint8_t *pal_idx;
 1211|  58.5k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1211:21): [True: 0, False: 58.5k]
  ------------------
 1212|      0|                    const int p = t->frame_thread.pass & 1;
 1213|      0|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1213:21): [True: 0, False: 0]
  ------------------
 1214|      0|                    pal_idx = ts->frame_thread[p].pal_idx;
 1215|      0|                    ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1216|  58.5k|                } else {
 1217|  58.5k|                    pal_idx = t->scratch.pal_idx_y;
 1218|  58.5k|                }
 1219|  58.5k|                const pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (1219:42): [True: 0, False: 58.5k]
  ------------------
 1220|      0|                    f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1221|      0|                                        ((t->bx >> 1) + (t->by & 1))][0] :
 1222|  58.5k|                    bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|  58.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  58.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1223|  58.5k|                f->dsp->ipred.pal_pred(dst, f->cur.stride[0], pal,
 1224|  58.5k|                                       pal_idx, bw4 * 4, bh4 * 4);
 1225|  58.5k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  58.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 58.5k]
  |  |  ------------------
  |  |   35|  58.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  58.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1226|      0|                    hex_dump(dst, PXSTRIDE(f->cur.stride[0]),
 1227|      0|                             bw4 * 4, bh4 * 4, "y-pal-pred");
 1228|  58.5k|            }
 1229|       |
 1230|  1.69M|            const int intra_flags = (sm_flag(t->a, bx4) |
 1231|  1.69M|                                     sm_flag(&t->l, by4) |
 1232|  1.69M|                                     intra_edge_filter_flag);
 1233|  1.69M|            const int sb_has_tr = init_x + 16 < w4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1233:35): [True: 137k, False: 1.55M]
  |  Branch (1233:58): [True: 67.7k, False: 1.49M]
  ------------------
 1234|  1.55M|                              intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT;
 1235|  1.69M|            const int sb_has_bl = init_x ? 0 : init_y + 16 < h4 ? 1 :
  ------------------
  |  Branch (1235:35): [True: 137k, False: 1.55M]
  |  Branch (1235:48): [True: 67.7k, False: 1.49M]
  ------------------
 1236|  1.55M|                              intra_edge_flags & EDGE_I444_LEFT_HAS_BOTTOM;
 1237|  1.69M|            int y, x;
 1238|  1.69M|            const int sub_w4 = imin(w4, init_x + 16);
 1239|  3.71M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (1239:47): [True: 2.02M, False: 1.69M]
  ------------------
 1240|  2.02M|                 y += t_dim->h, t->by += t_dim->h)
 1241|  2.02M|            {
 1242|  2.02M|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1243|  2.02M|                               4 * (t->by * PXSTRIDE(f->cur.stride[0]) +
 1244|  2.02M|                                    t->bx + init_x);
 1245|  5.56M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (1245:51): [True: 3.53M, False: 2.02M]
  ------------------
 1246|  3.53M|                     x += t_dim->w, t->bx += t_dim->w)
 1247|  3.53M|                {
 1248|  3.53M|                    if (b->pal_sz[0]) goto skip_y_pred;
  ------------------
  |  Branch (1248:25): [True: 89.4k, False: 3.45M]
  ------------------
 1249|       |
 1250|  3.45M|                    int angle = b->y_angle;
 1251|  3.45M|                    const enum EdgeFlags edge_flags =
 1252|  3.45M|                        (((y > init_y || !sb_has_tr) && (x + t_dim->w >= sub_w4)) ?
  ------------------
  |  Branch (1252:28): [True: 1.53M, False: 1.91M]
  |  Branch (1252:42): [True: 570k, False: 1.34M]
  |  Branch (1252:57): [True: 836k, False: 1.26M]
  ------------------
 1253|  2.61M|                             0 : EDGE_I444_TOP_HAS_RIGHT) |
 1254|  3.45M|                        ((x > init_x || (!sb_has_bl && y + t_dim->h >= sub_h4)) ?
  ------------------
  |  Branch (1254:27): [True: 1.49M, False: 1.95M]
  |  Branch (1254:42): [True: 1.12M, False: 827k]
  |  Branch (1254:56): [True: 893k, False: 236k]
  ------------------
 1255|  2.38M|                             0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1256|  3.45M|                    const pixel *top_sb_edge = NULL;
 1257|  3.45M|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1257:25): [True: 854k, False: 2.59M]
  ------------------
 1258|   854k|                        top_sb_edge = f->ipred_edge[0];
 1259|   854k|                        const int sby = t->by >> f->sb_shift;
 1260|   854k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1261|   854k|                    }
 1262|  3.45M|                    const enum IntraPredMode m =
 1263|  3.45M|                        bytefn(dav1d_prepare_intra_edges)(t->bx,
  ------------------
  |  |   87|  3.45M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  3.45M|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1264|  3.45M|                                                          t->bx > ts->tiling.col_start,
 1265|  3.45M|                                                          t->by,
 1266|  3.45M|                                                          t->by > ts->tiling.row_start,
 1267|  3.45M|                                                          ts->tiling.col_end,
 1268|  3.45M|                                                          ts->tiling.row_end,
 1269|  3.45M|                                                          edge_flags, dst,
 1270|  3.45M|                                                          f->cur.stride[0], top_sb_edge,
 1271|  3.45M|                                                          b->y_mode, &angle,
 1272|  3.45M|                                                          t_dim->w, t_dim->h,
 1273|  3.45M|                                                          f->seq_hdr->intra_edge_filter,
 1274|  3.45M|                                                          edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.45M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1275|  3.45M|                    dsp->ipred.intra_pred[m](dst, f->cur.stride[0], edge,
 1276|  3.45M|                                             t_dim->w * 4, t_dim->h * 4,
 1277|  3.45M|                                             angle | intra_flags,
 1278|  3.45M|                                             4 * f->bw - 4 * t->bx,
 1279|  3.45M|                                             4 * f->bh - 4 * t->by
 1280|  3.45M|                                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.45M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1281|       |
 1282|  3.45M|                    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  3.45M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 3.45M]
  |  |  ------------------
  |  |   35|  3.45M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  3.45M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1283|      0|                        hex_dump(edge - t_dim->h * 4, t_dim->h * 4,
 1284|      0|                                 t_dim->h * 4, 2, "l");
 1285|      0|                        hex_dump(edge, 0, 1, 1, "tl");
 1286|      0|                        hex_dump(edge + 1, t_dim->w * 4,
 1287|      0|                                 t_dim->w * 4, 2, "t");
 1288|      0|                        hex_dump(dst, f->cur.stride[0],
 1289|      0|                                 t_dim->w * 4, t_dim->h * 4, "y-intra-pred");
 1290|      0|                    }
 1291|       |
 1292|  3.53M|                skip_y_pred: {}
 1293|  3.53M|                    if (!b->skip) {
  ------------------
  |  Branch (1293:25): [True: 1.65M, False: 1.88M]
  ------------------
 1294|  1.65M|                        coef *cf;
 1295|  1.65M|                        int eob;
 1296|  1.65M|                        enum TxfmType txtp;
 1297|  1.65M|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1297:29): [True: 0, False: 1.65M]
  ------------------
 1298|      0|                            const int p = t->frame_thread.pass & 1;
 1299|      0|                            const int cbi = *ts->frame_thread[p].cbi++;
 1300|      0|                            cf = ts->frame_thread[p].cf;
 1301|      0|                            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
 1302|      0|                            eob  = cbi >> 5;
 1303|      0|                            txtp = cbi & 0x1f;
 1304|  1.65M|                        } else {
 1305|  1.65M|                            uint8_t cf_ctx;
 1306|  1.65M|                            cf = bitfn(t->cf);
  ------------------
  |  |   77|  1.65M|#define bitfn(x) x##_16bpc
  ------------------
 1307|  1.65M|                            eob = decode_coefs(t, &t->a->lcoef[bx4 + x],
 1308|  1.65M|                                               &t->l.lcoef[by4 + y], b->tx, bs,
 1309|  1.65M|                                               b, 1, 0, cf, &txtp, &cf_ctx);
 1310|  1.65M|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.65M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.65M]
  |  |  ------------------
  |  |   35|  1.65M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.65M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1311|      0|                                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
 1312|      0|                                       b->tx, txtp, eob, ts->msac.rng);
 1313|  1.65M|                            dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
 1314|  1.65M|                            dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
 1315|  1.65M|                        }
 1316|  1.65M|                        if (eob >= 0) {
  ------------------
  |  Branch (1316:29): [True: 1.24M, False: 406k]
  ------------------
 1317|  1.24M|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  1.24M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.24M]
  |  |  ------------------
  |  |   35|  1.24M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.24M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1318|      0|                                coef_dump(cf, imin(t_dim->h, 8) * 4,
 1319|      0|                                          imin(t_dim->w, 8) * 4, 3, "dq");
 1320|  1.24M|                            dsp->itx.itxfm_add[b->tx]
 1321|  1.24M|                                              [txtp](dst,
 1322|  1.24M|                                                     f->cur.stride[0],
 1323|  1.24M|                                                     cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  1.24M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1324|  1.24M|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  1.24M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.24M]
  |  |  ------------------
  |  |   35|  1.24M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.24M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1325|      0|                                hex_dump(dst, f->cur.stride[0],
 1326|      0|                                         t_dim->w * 4, t_dim->h * 4, "recon");
 1327|  1.24M|                        }
 1328|  1.88M|                    } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1328:32): [True: 1.88M, False: 0]
  ------------------
 1329|  1.88M|                        dav1d_memset_pow2[t_dim->lw](&t->a->lcoef[bx4 + x], 0x40);
 1330|  1.88M|                        dav1d_memset_pow2[t_dim->lh](&t->l.lcoef[by4 + y], 0x40);
 1331|  1.88M|                    }
 1332|  3.53M|                    dst += 4 * t_dim->w;
 1333|  3.53M|                }
 1334|  2.02M|                t->bx -= x;
 1335|  2.02M|            }
 1336|  1.69M|            t->by -= y;
 1337|       |
 1338|  1.69M|            if (!has_chroma) continue;
  ------------------
  |  Branch (1338:17): [True: 608k, False: 1.08M]
  ------------------
 1339|       |
 1340|  1.08M|            const ptrdiff_t stride = f->cur.stride[1];
 1341|       |
 1342|  1.08M|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1342:17): [True: 212k, False: 875k]
  ------------------
 1343|   212k|                assert(!init_x && !init_y);
  ------------------
  |  Branch (1343:17): [True: 212k, False: 0]
  |  Branch (1343:17): [True: 212k, False: 0]
  ------------------
 1344|       |
 1345|   212k|                int16_t *const ac = t->scratch.ac;
 1346|   212k|                pixel *y_src = ((pixel *) f->cur.data[0]) + 4 * (t->bx & ~ss_hor) +
 1347|   212k|                                 4 * (t->by & ~ss_ver) * PXSTRIDE(f->cur.stride[0]);
 1348|   212k|                const ptrdiff_t uv_off = 4 * ((t->bx >> ss_hor) +
 1349|   212k|                                              (t->by >> ss_ver) * PXSTRIDE(stride));
 1350|   212k|                pixel *const uv_dst[2] = { ((pixel *) f->cur.data[1]) + uv_off,
 1351|   212k|                                           ((pixel *) f->cur.data[2]) + uv_off };
 1352|       |
 1353|   212k|                const int furthest_r =
 1354|   212k|                    ((cw4 << ss_hor) + t_dim->w - 1) & ~(t_dim->w - 1);
 1355|   212k|                const int furthest_b =
 1356|   212k|                    ((ch4 << ss_ver) + t_dim->h - 1) & ~(t_dim->h - 1);
 1357|   212k|                dsp->ipred.cfl_ac[f->cur.p.layout - 1](ac, y_src, f->cur.stride[0],
 1358|   212k|                                                         cbw4 - (furthest_r >> ss_hor),
 1359|   212k|                                                         cbh4 - (furthest_b >> ss_ver),
 1360|   212k|                                                         cbw4 * 4, cbh4 * 4);
 1361|   638k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1361:34): [True: 425k, False: 212k]
  ------------------
 1362|   425k|                    if (!b->cfl_alpha[pl]) continue;
  ------------------
  |  Branch (1362:25): [True: 86.7k, False: 338k]
  ------------------
 1363|   338k|                    int angle = 0;
 1364|   338k|                    const pixel *top_sb_edge = NULL;
 1365|   338k|                    if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1365:25): [True: 106k, False: 232k]
  ------------------
 1366|   106k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1367|   106k|                        const int sby = t->by >> f->sb_shift;
 1368|   106k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1369|   106k|                    }
 1370|   338k|                    const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1371|   338k|                    const int xstart = ts->tiling.col_start >> ss_hor;
 1372|   338k|                    const int ystart = ts->tiling.row_start >> ss_ver;
 1373|   338k|                    const enum IntraPredMode m =
 1374|   338k|                        bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|   338k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   338k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1375|   338k|                                                          ypos, ypos > ystart,
 1376|   338k|                                                          ts->tiling.col_end >> ss_hor,
 1377|   338k|                                                          ts->tiling.row_end >> ss_ver,
 1378|   338k|                                                          0, uv_dst[pl], stride,
 1379|   338k|                                                          top_sb_edge, DC_PRED, &angle,
 1380|   338k|                                                          uv_t_dim->w, uv_t_dim->h, 0,
 1381|   338k|                                                          edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   338k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1382|   338k|                    dsp->ipred.cfl_pred[m](uv_dst[pl], stride, edge,
 1383|   338k|                                           uv_t_dim->w * 4,
 1384|   338k|                                           uv_t_dim->h * 4,
 1385|   338k|                                           ac, b->cfl_alpha[pl]
 1386|   338k|                                           HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   338k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1387|   338k|                }
 1388|   212k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|   212k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 212k]
  |  |  ------------------
  |  |   35|   212k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   212k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1389|      0|                    ac_dump(ac, 4*cbw4, 4*cbh4, "ac");
 1390|      0|                    hex_dump(uv_dst[0], stride, cbw4 * 4, cbh4 * 4, "u-cfl-pred");
 1391|      0|                    hex_dump(uv_dst[1], stride, cbw4 * 4, cbh4 * 4, "v-cfl-pred");
 1392|      0|                }
 1393|   875k|            } else if (b->pal_sz[1]) {
  ------------------
  |  Branch (1393:24): [True: 12.2k, False: 863k]
  ------------------
 1394|  12.2k|                const ptrdiff_t uv_dstoff = 4 * ((t->bx >> ss_hor) +
 1395|  12.2k|                                              (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
 1396|  12.2k|                const pixel (*pal)[8];
 1397|  12.2k|                const uint8_t *pal_idx;
 1398|  12.2k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1398:21): [True: 0, False: 12.2k]
  ------------------
 1399|      0|                    const int p = t->frame_thread.pass & 1;
 1400|      0|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1400:21): [True: 0, False: 0]
  ------------------
 1401|      0|                    pal = f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1402|      0|                                              ((t->bx >> 1) + (t->by & 1))];
 1403|      0|                    pal_idx = ts->frame_thread[p].pal_idx;
 1404|      0|                    ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1405|  12.2k|                } else {
 1406|  12.2k|                    pal = bytefn(t->scratch.pal);
  ------------------
  |  |   87|  12.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  12.2k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1407|  12.2k|                    pal_idx = t->scratch.pal_idx_uv;
 1408|  12.2k|                }
 1409|       |
 1410|  12.2k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[1]) + uv_dstoff,
 1411|  12.2k|                                       f->cur.stride[1], pal[1],
 1412|  12.2k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1413|  12.2k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[2]) + uv_dstoff,
 1414|  12.2k|                                       f->cur.stride[1], pal[2],
 1415|  12.2k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1416|  12.2k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  12.2k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 12.2k]
  |  |  ------------------
  |  |   35|  12.2k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  12.2k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1417|      0|                    hex_dump(((pixel *) f->cur.data[1]) + uv_dstoff,
 1418|      0|                             PXSTRIDE(f->cur.stride[1]),
 1419|      0|                             cbw4 * 4, cbh4 * 4, "u-pal-pred");
 1420|      0|                    hex_dump(((pixel *) f->cur.data[2]) + uv_dstoff,
 1421|      0|                             PXSTRIDE(f->cur.stride[1]),
 1422|      0|                             cbw4 * 4, cbh4 * 4, "v-pal-pred");
 1423|      0|                }
 1424|  12.2k|            }
 1425|       |
 1426|  1.08M|            const int sm_uv_fl = sm_uv_flag(t->a, cbx4) |
 1427|  1.08M|                                 sm_uv_flag(&t->l, cby4);
 1428|  1.08M|            const int uv_sb_has_tr =
 1429|  1.08M|                ((init_x + 16) >> ss_hor) < cw4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1429:17): [True: 77.0k, False: 1.01M]
  |  Branch (1429:55): [True: 38.6k, False: 972k]
  ------------------
 1430|  1.01M|                intra_edge_flags & (EDGE_I420_TOP_HAS_RIGHT >> (f->cur.p.layout - 1));
 1431|  1.08M|            const int uv_sb_has_bl =
 1432|  1.08M|                init_x ? 0 : ((init_y + 16) >> ss_ver) < ch4 ? 1 :
  ------------------
  |  Branch (1432:17): [True: 77.0k, False: 1.01M]
  |  Branch (1432:30): [True: 38.6k, False: 972k]
  ------------------
 1433|  1.01M|                intra_edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM >> (f->cur.p.layout - 1));
 1434|  1.08M|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
 1435|  3.26M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1435:30): [True: 2.17M, False: 1.08M]
  ------------------
 1436|  4.66M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (1436:61): [True: 2.48M, False: 2.17M]
  ------------------
 1437|  2.48M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
 1438|  2.48M|                {
 1439|  2.48M|                    pixel *dst = ((pixel *) f->cur.data[1 + pl]) +
 1440|  2.48M|                                   4 * ((t->by >> ss_ver) * PXSTRIDE(stride) +
 1441|  2.48M|                                        ((t->bx + init_x) >> ss_hor));
 1442|  6.12M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (1442:65): [True: 3.63M, False: 2.48M]
  ------------------
 1443|  3.63M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
 1444|  3.63M|                    {
 1445|  3.63M|                        if ((b->uv_mode == CFL_PRED && b->cfl_alpha[pl]) ||
  ------------------
  |  Branch (1445:30): [True: 425k, False: 3.21M]
  |  Branch (1445:56): [True: 338k, False: 86.7k]
  ------------------
 1446|  3.29M|                            b->pal_sz[1])
  ------------------
  |  Branch (1446:29): [True: 32.7k, False: 3.26M]
  ------------------
 1447|   371k|                        {
 1448|   371k|                            goto skip_uv_pred;
 1449|   371k|                        }
 1450|       |
 1451|  3.26M|                        int angle = b->uv_angle;
 1452|       |                        // this probably looks weird because we're using
 1453|       |                        // luma flags in a chroma loop, but that's because
 1454|       |                        // prepare_intra_edges() expects luma flags as input
 1455|  3.26M|                        const enum EdgeFlags edge_flags =
 1456|  3.26M|                            (((y > (init_y >> ss_ver) || !uv_sb_has_tr) &&
  ------------------
  |  Branch (1456:32): [True: 1.18M, False: 2.07M]
  |  Branch (1456:58): [True: 688k, False: 1.39M]
  ------------------
 1457|  1.87M|                              (x + uv_t_dim->w >= sub_cw4)) ?
  ------------------
  |  Branch (1457:31): [True: 949k, False: 925k]
  ------------------
 1458|  2.31M|                                 0 : EDGE_I444_TOP_HAS_RIGHT) |
 1459|  3.26M|                            ((x > (init_x >> ss_hor) ||
  ------------------
  |  Branch (1459:31): [True: 1.14M, False: 2.12M]
  ------------------
 1460|  2.12M|                              (!uv_sb_has_bl && y + uv_t_dim->h >= sub_ch4)) ?
  ------------------
  |  Branch (1460:32): [True: 1.20M, False: 914k]
  |  Branch (1460:49): [True: 969k, False: 238k]
  ------------------
 1461|  2.11M|                                 0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1462|  3.26M|                        const pixel *top_sb_edge = NULL;
 1463|  3.26M|                        if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1463:29): [True: 809k, False: 2.45M]
  ------------------
 1464|   809k|                            top_sb_edge = f->ipred_edge[1 + pl];
 1465|   809k|                            const int sby = t->by >> f->sb_shift;
 1466|   809k|                            top_sb_edge += f->sb128w * 128 * (sby - 1);
 1467|   809k|                        }
 1468|  3.26M|                        const enum IntraPredMode uv_mode =
 1469|  3.26M|                             b->uv_mode == CFL_PRED ? DC_PRED : b->uv_mode;
  ------------------
  |  Branch (1469:30): [True: 86.7k, False: 3.17M]
  ------------------
 1470|  3.26M|                        const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1471|  3.26M|                        const int xstart = ts->tiling.col_start >> ss_hor;
 1472|  3.26M|                        const int ystart = ts->tiling.row_start >> ss_ver;
 1473|  3.26M|                        const enum IntraPredMode m =
 1474|  3.26M|                            bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  3.26M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  3.26M|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1475|  3.26M|                                                              ypos, ypos > ystart,
 1476|  3.26M|                                                              ts->tiling.col_end >> ss_hor,
 1477|  3.26M|                                                              ts->tiling.row_end >> ss_ver,
 1478|  3.26M|                                                              edge_flags, dst, stride,
 1479|  3.26M|                                                              top_sb_edge, uv_mode,
 1480|  3.26M|                                                              &angle, uv_t_dim->w,
 1481|  3.26M|                                                              uv_t_dim->h,
 1482|  3.26M|                                                              f->seq_hdr->intra_edge_filter,
 1483|  3.26M|                                                              edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.26M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1484|  3.26M|                        angle |= intra_edge_filter_flag;
 1485|  3.26M|                        dsp->ipred.intra_pred[m](dst, stride, edge,
 1486|  3.26M|                                                 uv_t_dim->w * 4,
 1487|  3.26M|                                                 uv_t_dim->h * 4,
 1488|  3.26M|                                                 angle | sm_uv_fl,
 1489|  3.26M|                                                 (4 * f->bw + ss_hor -
 1490|  3.26M|                                                  4 * (t->bx & ~ss_hor)) >> ss_hor,
 1491|  3.26M|                                                 (4 * f->bh + ss_ver -
 1492|  3.26M|                                                  4 * (t->by & ~ss_ver)) >> ss_ver
 1493|  3.26M|                                                 HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.26M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1494|  3.26M|                        if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  3.26M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 3.26M]
  |  |  ------------------
  |  |   35|  3.26M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  3.26M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                      if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1495|      0|                            hex_dump(edge - uv_t_dim->h * 4, uv_t_dim->h * 4,
 1496|      0|                                     uv_t_dim->h * 4, 2, "l");
 1497|      0|                            hex_dump(edge, 0, 1, 1, "tl");
 1498|      0|                            hex_dump(edge + 1, uv_t_dim->w * 4,
 1499|      0|                                     uv_t_dim->w * 4, 2, "t");
 1500|      0|                            hex_dump(dst, stride, uv_t_dim->w * 4,
 1501|      0|                                     uv_t_dim->h * 4, pl ? "v-intra-pred" : "u-intra-pred");
  ------------------
  |  Branch (1501:55): [True: 0, False: 0]
  ------------------
 1502|      0|                        }
 1503|       |
 1504|  3.63M|                    skip_uv_pred: {}
 1505|  3.63M|                        if (!b->skip) {
  ------------------
  |  Branch (1505:29): [True: 1.86M, False: 1.76M]
  ------------------
 1506|  1.86M|                            enum TxfmType txtp;
 1507|  1.86M|                            int eob;
 1508|  1.86M|                            coef *cf;
 1509|  1.86M|                            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1509:33): [True: 0, False: 1.86M]
  ------------------
 1510|      0|                                const int p = t->frame_thread.pass & 1;
 1511|      0|                                const int cbi = *ts->frame_thread[p].cbi++;
 1512|      0|                                cf = ts->frame_thread[p].cf;
 1513|      0|                                ts->frame_thread[p].cf += uv_t_dim->w * uv_t_dim->h * 16;
 1514|      0|                                eob  = cbi >> 5;
 1515|      0|                                txtp = cbi & 0x1f;
 1516|  1.86M|                            } else {
 1517|  1.86M|                                uint8_t cf_ctx;
 1518|  1.86M|                                cf = bitfn(t->cf);
  ------------------
  |  |   77|  1.86M|#define bitfn(x) x##_16bpc
  ------------------
 1519|  1.86M|                                eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1520|  1.86M|                                                   &t->l.ccoef[pl][cby4 + y],
 1521|  1.86M|                                                   b->uvtx, bs, b, 1, 1 + pl, cf,
 1522|  1.86M|                                                   &txtp, &cf_ctx);
 1523|  1.86M|                                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.86M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.86M]
  |  |  ------------------
  |  |   35|  1.86M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.86M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1524|      0|                                    printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1525|      0|                                           "txtp=%d,eob=%d]: r=%d [x=%d,cbx4=%d]\n",
 1526|      0|                                           pl, b->uvtx, txtp, eob, ts->msac.rng, x, cbx4);
 1527|  1.86M|                                int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1528|  1.86M|                                int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1529|  1.86M|                                dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1530|  1.86M|                                dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1531|  1.86M|                            }
 1532|  1.86M|                            if (eob >= 0) {
  ------------------
  |  Branch (1532:33): [True: 512k, False: 1.35M]
  ------------------
 1533|   512k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   512k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 512k]
  |  |  ------------------
  |  |   35|   512k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   512k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1534|      0|                                    coef_dump(cf, uv_t_dim->h * 4,
 1535|      0|                                              uv_t_dim->w * 4, 3, "dq");
 1536|   512k|                                dsp->itx.itxfm_add[b->uvtx]
 1537|   512k|                                                  [txtp](dst, stride,
 1538|   512k|                                                         cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   512k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1539|   512k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   512k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 512k]
  |  |  ------------------
  |  |   35|   512k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   512k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1540|      0|                                    hex_dump(dst, stride, uv_t_dim->w * 4,
 1541|      0|                                             uv_t_dim->h * 4, "recon");
 1542|   512k|                            }
 1543|  1.86M|                        } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1543:36): [True: 1.76M, False: 0]
  ------------------
 1544|  1.76M|                            dav1d_memset_pow2[uv_t_dim->lw](&t->a->ccoef[pl][cbx4 + x], 0x40);
 1545|  1.76M|                            dav1d_memset_pow2[uv_t_dim->lh](&t->l.ccoef[pl][cby4 + y], 0x40);
 1546|  1.76M|                        }
 1547|  3.63M|                        dst += uv_t_dim->w * 4;
 1548|  3.63M|                    }
 1549|  2.48M|                    t->bx -= x << ss_hor;
 1550|  2.48M|                }
 1551|  2.17M|                t->by -= y << ss_ver;
 1552|  2.17M|            }
 1553|  1.08M|        }
 1554|  1.55M|    }
 1555|  1.49M|}
dav1d_recon_b_inter_16bpc:
 1559|  1.13M|{
 1560|  1.13M|    Dav1dTileState *const ts = t->ts;
 1561|  1.13M|    const Dav1dFrameContext *const f = t->f;
 1562|  1.13M|    const Dav1dDSPContext *const dsp = f->dsp;
 1563|  1.13M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1564|  1.13M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1565|  1.13M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1566|  1.13M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1567|  1.13M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1568|  1.13M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1569|  1.13M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1570|  1.13M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1570:28): [True: 418k, False: 716k]
  ------------------
 1571|   418k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1571:29): [True: 380k, False: 37.8k]
  |  Branch (1571:45): [True: 18.8k, False: 18.9k]
  ------------------
 1572|   399k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1572:29): [True: 377k, False: 21.5k]
  |  Branch (1572:45): [True: 10.3k, False: 11.2k]
  ------------------
 1573|  1.13M|    const int chr_layout_idx = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I400 ? 0 :
  ------------------
  |  Branch (1573:32): [True: 716k, False: 418k]
  ------------------
 1574|  1.13M|                               DAV1D_PIXEL_LAYOUT_I444 - f->cur.p.layout;
 1575|  1.13M|    int res;
 1576|       |
 1577|       |    // prediction
 1578|  1.13M|    const int cbh4 = (bh4 + ss_ver) >> ss_ver, cbw4 = (bw4 + ss_hor) >> ss_hor;
 1579|  1.13M|    pixel *dst = ((pixel *) f->cur.data[0]) +
 1580|  1.13M|        4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
 1581|  1.13M|    const ptrdiff_t uvdstoff =
 1582|  1.13M|        4 * ((t->bx >> ss_hor) + (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
 1583|  1.13M|    if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  1.13M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  1.13M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 506k, False: 628k]
  |  |  ------------------
  ------------------
 1584|       |        // intrabc
 1585|   506k|        assert(!f->frame_hdr->super_res.enabled);
  ------------------
  |  Branch (1585:9): [True: 506k, False: 0]
  ------------------
 1586|   506k|        res = mc(t, dst, NULL, f->cur.stride[0], bw4, bh4, t->bx, t->by, 0,
 1587|   506k|                 b->mv[0], &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1588|   506k|        if (res) return res;
  ------------------
  |  Branch (1588:13): [True: 0, False: 506k]
  ------------------
 1589|   506k|        if (has_chroma) for (int pl = 1; pl < 3; pl++) {
  ------------------
  |  Branch (1589:13): [True: 133k, False: 372k]
  |  Branch (1589:42): [True: 267k, False: 133k]
  ------------------
 1590|   267k|            res = mc(t, ((pixel *)f->cur.data[pl]) + uvdstoff, NULL, f->cur.stride[1],
 1591|   267k|                     bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1592|   267k|                     t->bx & ~ss_hor, t->by & ~ss_ver, pl, b->mv[0],
 1593|   267k|                     &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1594|   267k|            if (res) return res;
  ------------------
  |  Branch (1594:17): [True: 0, False: 267k]
  ------------------
 1595|   267k|        }
 1596|   628k|    } else if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1596:16): [True: 528k, False: 100k]
  ------------------
 1597|   528k|        const Dav1dThreadPicture *const refp = &f->refp[b->ref[0]];
 1598|   528k|        const enum Filter2d filter_2d = b->filter2d;
 1599|       |
 1600|   528k|        if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1600:13): [True: 323k, False: 204k]
  ------------------
 1601|   323k|            ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1601:15): [True: 200k, False: 122k]
  |  Branch (1601:44): [True: 12.9k, False: 187k]
  ------------------
 1602|   310k|             (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1602:15): [True: 47.4k, False: 263k]
  |  Branch (1602:44): [True: 44.9k, False: 2.49k]
  ------------------
 1603|  57.9k|        {
 1604|  57.9k|            res = warp_affine(t, dst, NULL, f->cur.stride[0], b_dim, 0, refp,
 1605|  57.9k|                              b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1605:31): [True: 44.9k, False: 12.9k]
  ------------------
 1606|  57.9k|                                  &f->frame_hdr->gmv[b->ref[0]]);
 1607|  57.9k|            if (res) return res;
  ------------------
  |  Branch (1607:17): [True: 0, False: 57.9k]
  ------------------
 1608|   470k|        } else {
 1609|   470k|            res = mc(t, dst, NULL, f->cur.stride[0],
 1610|   470k|                     bw4, bh4, t->bx, t->by, 0, b->mv[0], refp, b->ref[0], filter_2d);
 1611|   470k|            if (res) return res;
  ------------------
  |  Branch (1611:17): [True: 0, False: 470k]
  ------------------
 1612|   470k|            if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1612:17): [True: 105k, False: 364k]
  ------------------
 1613|   105k|                res = obmc(t, dst, f->cur.stride[0], b_dim, 0, bx4, by4, w4, h4);
 1614|   105k|                if (res) return res;
  ------------------
  |  Branch (1614:21): [True: 0, False: 105k]
  ------------------
 1615|   105k|            }
 1616|   470k|        }
 1617|   528k|        if (b->interintra_type) {
  ------------------
  |  Branch (1617:13): [True: 21.7k, False: 506k]
  ------------------
 1618|  21.7k|            pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   77|  21.7k|#define bitfn(x) x##_16bpc
  ------------------
 1619|  21.7k|            enum IntraPredMode m = b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1619:36): [True: 3.88k, False: 17.8k]
  ------------------
 1620|  17.8k|                                   SMOOTH_PRED : b->interintra_mode;
 1621|  21.7k|            pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   77|  21.7k|#define bitfn(x) x##_16bpc
  ------------------
 1622|  21.7k|            int angle = 0;
 1623|  21.7k|            const pixel *top_sb_edge = NULL;
 1624|  21.7k|            if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1624:17): [True: 3.86k, False: 17.8k]
  ------------------
 1625|  3.86k|                top_sb_edge = f->ipred_edge[0];
 1626|  3.86k|                const int sby = t->by >> f->sb_shift;
 1627|  3.86k|                top_sb_edge += f->sb128w * 128 * (sby - 1);
 1628|  3.86k|            }
 1629|  21.7k|            m = bytefn(dav1d_prepare_intra_edges)(t->bx, t->bx > ts->tiling.col_start,
  ------------------
  |  |   87|  21.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  21.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1630|  21.7k|                                                  t->by, t->by > ts->tiling.row_start,
 1631|  21.7k|                                                  ts->tiling.col_end, ts->tiling.row_end,
 1632|  21.7k|                                                  0, dst, f->cur.stride[0], top_sb_edge,
 1633|  21.7k|                                                  m, &angle, bw4, bh4, 0, tl_edge
 1634|  21.7k|                                                  HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  21.7k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1635|  21.7k|            dsp->ipred.intra_pred[m](tmp, 4 * bw4 * sizeof(pixel),
 1636|  21.7k|                                     tl_edge, bw4 * 4, bh4 * 4, 0, 0, 0
 1637|  21.7k|                                     HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  21.7k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1638|  21.7k|            dsp->mc.blend(dst, f->cur.stride[0], tmp,
 1639|  21.7k|                          bw4 * 4, bh4 * 4, II_MASK(0, bs, b));
  ------------------
  |  |   83|  21.7k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  21.7k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 16.8k, False: 4.82k]
  |  |  ------------------
  |  |   85|  21.7k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  21.7k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1640|  21.7k|        }
 1641|       |
 1642|   528k|        if (!has_chroma) goto skip_inter_chroma_pred;
  ------------------
  |  Branch (1642:13): [True: 343k, False: 184k]
  ------------------
 1643|       |
 1644|       |        // sub8x8 derivation
 1645|   184k|        int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1645:25): [True: 7.20k, False: 177k]
  |  Branch (1645:42): [True: 2.67k, False: 174k]
  ------------------
 1646|   184k|        refmvs_block *const *r;
 1647|   184k|        if (is_sub8x8) {
  ------------------
  |  Branch (1647:13): [True: 9.87k, False: 174k]
  ------------------
 1648|  9.87k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1648:13): [True: 9.87k, False: 0]
  ------------------
 1649|  9.87k|            r = &t->rt.r[(t->by & 31) + 5];
 1650|  9.87k|            if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1650:17): [True: 7.20k, False: 2.67k]
  ------------------
 1651|  9.87k|            if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1651:17): [True: 7.31k, False: 2.56k]
  ------------------
 1652|  9.87k|            if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1652:17): [True: 7.20k, False: 2.67k]
  |  Branch (1652:29): [True: 4.64k, False: 2.56k]
  ------------------
 1653|  4.64k|                is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1654|  9.87k|        }
 1655|       |
 1656|       |        // chroma prediction
 1657|   184k|        if (is_sub8x8) {
  ------------------
  |  Branch (1657:13): [True: 9.39k, False: 175k]
  ------------------
 1658|  9.39k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1658:13): [True: 9.39k, False: 0]
  ------------------
 1659|  9.39k|            ptrdiff_t h_off = 0, v_off = 0;
 1660|  9.39k|            if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1660:17): [True: 6.81k, False: 2.58k]
  |  Branch (1660:29): [True: 4.29k, False: 2.51k]
  ------------------
 1661|  12.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1661:34): [True: 8.58k, False: 4.29k]
  ------------------
 1662|  8.58k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1663|  8.58k|                             NULL, f->cur.stride[1],
 1664|  8.58k|                             bw4, bh4, t->bx - 1, t->by - 1, 1 + pl,
 1665|  8.58k|                             r[-1][t->bx - 1].mv.mv[0],
 1666|  8.58k|                             &f->refp[r[-1][t->bx - 1].ref.ref[0] - 1],
 1667|  8.58k|                             r[-1][t->bx - 1].ref.ref[0] - 1,
 1668|  8.58k|                             t->frame_thread.pass != 2 ? t->tl_4x4_filter :
  ------------------
  |  Branch (1668:30): [True: 8.58k, False: 0]
  ------------------
 1669|  8.58k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx - 1].filter2d);
 1670|  8.58k|                    if (res) return res;
  ------------------
  |  Branch (1670:25): [True: 0, False: 8.58k]
  ------------------
 1671|  8.58k|                }
 1672|  4.29k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
 1673|  4.29k|                h_off = 2;
 1674|  4.29k|            }
 1675|  9.39k|            if (bw4 == 1) {
  ------------------
  |  Branch (1675:17): [True: 6.81k, False: 2.58k]
  ------------------
 1676|  6.81k|                const enum Filter2d left_filter_2d =
 1677|  6.81k|                    dav1d_filter_2d[t->l.filter[1][by4]][t->l.filter[0][by4]];
 1678|  20.4k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1678:34): [True: 13.6k, False: 6.81k]
  ------------------
 1679|  13.6k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + v_off, NULL,
 1680|  13.6k|                             f->cur.stride[1], bw4, bh4, t->bx - 1,
 1681|  13.6k|                             t->by, 1 + pl, r[0][t->bx - 1].mv.mv[0],
 1682|  13.6k|                             &f->refp[r[0][t->bx - 1].ref.ref[0] - 1],
 1683|  13.6k|                             r[0][t->bx - 1].ref.ref[0] - 1,
 1684|  13.6k|                             t->frame_thread.pass != 2 ? left_filter_2d :
  ------------------
  |  Branch (1684:30): [True: 13.6k, False: 0]
  ------------------
 1685|  13.6k|                                 f->frame_thread.b[(t->by * f->b4_stride) + t->bx - 1].filter2d);
 1686|  13.6k|                    if (res) return res;
  ------------------
  |  Branch (1686:25): [True: 0, False: 13.6k]
  ------------------
 1687|  13.6k|                }
 1688|  6.81k|                h_off = 2;
 1689|  6.81k|            }
 1690|  9.39k|            if (bh4 == ss_ver) {
  ------------------
  |  Branch (1690:17): [True: 6.88k, False: 2.51k]
  ------------------
 1691|  6.88k|                const enum Filter2d top_filter_2d =
 1692|  6.88k|                    dav1d_filter_2d[t->a->filter[1][bx4]][t->a->filter[0][bx4]];
 1693|  20.6k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1693:34): [True: 13.7k, False: 6.88k]
  ------------------
 1694|  13.7k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off, NULL,
 1695|  13.7k|                             f->cur.stride[1], bw4, bh4, t->bx, t->by - 1,
 1696|  13.7k|                             1 + pl, r[-1][t->bx].mv.mv[0],
 1697|  13.7k|                             &f->refp[r[-1][t->bx].ref.ref[0] - 1],
 1698|  13.7k|                             r[-1][t->bx].ref.ref[0] - 1,
 1699|  13.7k|                             t->frame_thread.pass != 2 ? top_filter_2d :
  ------------------
  |  Branch (1699:30): [True: 13.7k, False: 0]
  ------------------
 1700|  13.7k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx].filter2d);
 1701|  13.7k|                    if (res) return res;
  ------------------
  |  Branch (1701:25): [True: 0, False: 13.7k]
  ------------------
 1702|  13.7k|                }
 1703|  6.88k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
 1704|  6.88k|            }
 1705|  28.1k|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1705:30): [True: 18.7k, False: 9.39k]
  ------------------
 1706|  18.7k|                res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off + v_off, NULL, f->cur.stride[1],
 1707|  18.7k|                         bw4, bh4, t->bx, t->by, 1 + pl, b->mv[0],
 1708|  18.7k|                         refp, b->ref[0], filter_2d);
 1709|  18.7k|                if (res) return res;
  ------------------
  |  Branch (1709:21): [True: 0, False: 18.7k]
  ------------------
 1710|  18.7k|            }
 1711|   175k|        } else {
 1712|   175k|            if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (1712:17): [True: 101k, False: 73.2k]
  ------------------
 1713|   101k|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1713:19): [True: 50.2k, False: 51.7k]
  |  Branch (1713:48): [True: 8.24k, False: 42.0k]
  ------------------
 1714|  93.7k|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1714:19): [True: 4.84k, False: 88.8k]
  |  Branch (1714:48): [True: 3.95k, False: 891]
  ------------------
 1715|  12.2k|            {
 1716|  36.6k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1716:34): [True: 24.4k, False: 12.2k]
  ------------------
 1717|  24.4k|                    res = warp_affine(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff, NULL,
 1718|  24.4k|                                      f->cur.stride[1], b_dim, 1 + pl, refp,
 1719|  24.4k|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1719:39): [True: 7.91k, False: 16.4k]
  ------------------
 1720|  24.4k|                                          &f->frame_hdr->gmv[b->ref[0]]);
 1721|  24.4k|                    if (res) return res;
  ------------------
  |  Branch (1721:25): [True: 0, False: 24.4k]
  ------------------
 1722|  24.4k|                }
 1723|   163k|            } else {
 1724|   489k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1724:34): [True: 326k, False: 163k]
  ------------------
 1725|   326k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1726|   326k|                             NULL, f->cur.stride[1],
 1727|   326k|                             bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1728|   326k|                             t->bx & ~ss_hor, t->by & ~ss_ver,
 1729|   326k|                             1 + pl, b->mv[0], refp, b->ref[0], filter_2d);
 1730|   326k|                    if (res) return res;
  ------------------
  |  Branch (1730:25): [True: 0, False: 326k]
  ------------------
 1731|   326k|                    if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1731:25): [True: 66.5k, False: 259k]
  ------------------
 1732|  66.5k|                        res = obmc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1733|  66.5k|                                   f->cur.stride[1], b_dim, 1 + pl, bx4, by4, w4, h4);
 1734|  66.5k|                        if (res) return res;
  ------------------
  |  Branch (1734:29): [True: 0, False: 66.5k]
  ------------------
 1735|  66.5k|                    }
 1736|   326k|                }
 1737|   163k|            }
 1738|   175k|            if (b->interintra_type) {
  ------------------
  |  Branch (1738:17): [True: 12.6k, False: 162k]
  ------------------
 1739|       |                // FIXME for 8x32 with 4:2:2 subsampling, this probably does
 1740|       |                // the wrong thing since it will select 4x16, not 4x32, as a
 1741|       |                // transform size...
 1742|  12.6k|                const uint8_t *const ii_mask = II_MASK(chr_layout_idx, bs, b);
  ------------------
  |  |   83|  12.6k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  12.6k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 9.77k, False: 2.83k]
  |  |  ------------------
  |  |   85|  12.6k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  12.6k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1743|       |
 1744|  37.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1744:34): [True: 25.2k, False: 12.6k]
  ------------------
 1745|  25.2k|                    pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   77|  25.2k|#define bitfn(x) x##_16bpc
  ------------------
 1746|  25.2k|                    pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   77|  25.2k|#define bitfn(x) x##_16bpc
  ------------------
 1747|  25.2k|                    enum IntraPredMode m =
 1748|  25.2k|                        b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1748:25): [True: 4.67k, False: 20.5k]
  ------------------
 1749|  20.5k|                        SMOOTH_PRED : b->interintra_mode;
 1750|  25.2k|                    int angle = 0;
 1751|  25.2k|                    pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1752|  25.2k|                    const pixel *top_sb_edge = NULL;
 1753|  25.2k|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1753:25): [True: 5.66k, False: 19.5k]
  ------------------
 1754|  5.66k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1755|  5.66k|                        const int sby = t->by >> f->sb_shift;
 1756|  5.66k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1757|  5.66k|                    }
 1758|  25.2k|                    m = bytefn(dav1d_prepare_intra_edges)(t->bx >> ss_hor,
  ------------------
  |  |   87|  25.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  25.2k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1759|  25.2k|                                                          (t->bx >> ss_hor) >
 1760|  25.2k|                                                              (ts->tiling.col_start >> ss_hor),
 1761|  25.2k|                                                          t->by >> ss_ver,
 1762|  25.2k|                                                          (t->by >> ss_ver) >
 1763|  25.2k|                                                              (ts->tiling.row_start >> ss_ver),
 1764|  25.2k|                                                          ts->tiling.col_end >> ss_hor,
 1765|  25.2k|                                                          ts->tiling.row_end >> ss_ver,
 1766|  25.2k|                                                          0, uvdst, f->cur.stride[1],
 1767|  25.2k|                                                          top_sb_edge, m,
 1768|  25.2k|                                                          &angle, cbw4, cbh4, 0, tl_edge
 1769|  25.2k|                                                          HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  25.2k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1770|  25.2k|                    dsp->ipred.intra_pred[m](tmp, cbw4 * 4 * sizeof(pixel),
 1771|  25.2k|                                             tl_edge, cbw4 * 4, cbh4 * 4, 0, 0, 0
 1772|  25.2k|                                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  25.2k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1773|  25.2k|                    dsp->mc.blend(uvdst, f->cur.stride[1], tmp,
 1774|  25.2k|                                  cbw4 * 4, cbh4 * 4, ii_mask);
 1775|  25.2k|                }
 1776|  12.6k|            }
 1777|   175k|        }
 1778|       |
 1779|   528k|    skip_inter_chroma_pred: {}
 1780|   528k|        t->tl_4x4_filter = filter_2d;
 1781|   528k|    } else {
 1782|   100k|        const enum Filter2d filter_2d = b->filter2d;
 1783|       |        // Maximum super block size is 128x128
 1784|   100k|        int16_t (*tmp)[128 * 128] = t->scratch.compinter;
 1785|   100k|        int jnt_weight;
 1786|   100k|        uint8_t *const seg_mask = t->scratch.seg_mask;
 1787|   100k|        const uint8_t *mask;
 1788|       |
 1789|   300k|        for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1789:25): [True: 200k, False: 100k]
  ------------------
 1790|   200k|            const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1791|       |
 1792|   200k|            if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (1792:17): [True: 17.5k, False: 182k]
  |  Branch (1792:55): [True: 854, False: 16.6k]
  ------------------
 1793|    854|                res = warp_affine(t, NULL, tmp[i], bw4 * 4, b_dim, 0, refp,
 1794|    854|                                  &f->frame_hdr->gmv[b->ref[i]]);
 1795|    854|                if (res) return res;
  ------------------
  |  Branch (1795:21): [True: 0, False: 854]
  ------------------
 1796|   199k|            } else {
 1797|   199k|                res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by, 0,
 1798|   199k|                         b->mv[i], refp, b->ref[i], filter_2d);
 1799|   199k|                if (res) return res;
  ------------------
  |  Branch (1799:21): [True: 0, False: 199k]
  ------------------
 1800|   199k|            }
 1801|   200k|        }
 1802|   100k|        switch (b->comp_type) {
  ------------------
  |  Branch (1802:17): [True: 100k, False: 0]
  ------------------
 1803|  49.6k|        case COMP_INTER_AVG:
  ------------------
  |  Branch (1803:9): [True: 49.6k, False: 50.5k]
  ------------------
 1804|  49.6k|            dsp->mc.avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1805|  49.6k|                        bw4 * 4, bh4 * 4 HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  49.6k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1806|  49.6k|            break;
 1807|  17.4k|        case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1807:9): [True: 17.4k, False: 82.7k]
  ------------------
 1808|  17.4k|            jnt_weight = f->jnt_weights[b->ref[0]][b->ref[1]];
 1809|  17.4k|            dsp->mc.w_avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1810|  17.4k|                          bw4 * 4, bh4 * 4, jnt_weight HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  17.4k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1811|  17.4k|            break;
 1812|  22.1k|        case COMP_INTER_SEG:
  ------------------
  |  Branch (1812:9): [True: 22.1k, False: 77.9k]
  ------------------
 1813|  22.1k|            dsp->mc.w_mask[chr_layout_idx](dst, f->cur.stride[0],
 1814|  22.1k|                                           tmp[b->mask_sign], tmp[!b->mask_sign],
 1815|  22.1k|                                           bw4 * 4, bh4 * 4, seg_mask,
 1816|  22.1k|                                           b->mask_sign HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  22.1k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1817|  22.1k|            mask = seg_mask;
 1818|  22.1k|            break;
 1819|  10.8k|        case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1819:9): [True: 10.8k, False: 89.2k]
  ------------------
 1820|  10.8k|            mask = WEDGE_MASK(0, bs, 0, b->wedge_idx);
  ------------------
  |  |   89|  10.8k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  10.8k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1821|  10.8k|            dsp->mc.mask(dst, f->cur.stride[0],
 1822|  10.8k|                         tmp[b->mask_sign], tmp[!b->mask_sign],
 1823|  10.8k|                         bw4 * 4, bh4 * 4, mask HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  10.8k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1824|  10.8k|            if (has_chroma)
  ------------------
  |  Branch (1824:17): [True: 8.06k, False: 2.83k]
  ------------------
 1825|  8.06k|                mask = WEDGE_MASK(chr_layout_idx, bs, b->mask_sign, b->wedge_idx);
  ------------------
  |  |   89|  8.06k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  8.06k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1826|  10.8k|            break;
 1827|   100k|        }
 1828|       |
 1829|       |        // chroma
 1830|   209k|        if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1830:13): [True: 69.8k, False: 30.3k]
  |  Branch (1830:42): [True: 139k, False: 69.8k]
  ------------------
 1831|   418k|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1831:29): [True: 279k, False: 139k]
  ------------------
 1832|   279k|                const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1833|   279k|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (1833:21): [True: 26.0k, False: 253k]
  ------------------
 1834|  26.0k|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (1834:21): [True: 24.7k, False: 1.34k]
  |  Branch (1834:45): [True: 738, False: 23.9k]
  ------------------
 1835|    738|                {
 1836|    738|                    res = warp_affine(t, NULL, tmp[i], bw4 * 4 >> ss_hor,
 1837|    738|                                      b_dim, 1 + pl,
 1838|    738|                                      refp, &f->frame_hdr->gmv[b->ref[i]]);
 1839|    738|                    if (res) return res;
  ------------------
  |  Branch (1839:25): [True: 0, False: 738]
  ------------------
 1840|   278k|                } else {
 1841|   278k|                    res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by,
 1842|   278k|                             1 + pl, b->mv[i], refp, b->ref[i], filter_2d);
 1843|   278k|                    if (res) return res;
  ------------------
  |  Branch (1843:25): [True: 0, False: 278k]
  ------------------
 1844|   278k|                }
 1845|   279k|            }
 1846|   139k|            pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1847|   139k|            switch (b->comp_type) {
  ------------------
  |  Branch (1847:21): [True: 139k, False: 0]
  ------------------
 1848|  68.1k|            case COMP_INTER_AVG:
  ------------------
  |  Branch (1848:13): [True: 68.1k, False: 71.4k]
  ------------------
 1849|  68.1k|                dsp->mc.avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1850|  68.1k|                            bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver
 1851|  68.1k|                            HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  68.1k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1852|  68.1k|                break;
 1853|  28.4k|            case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1853:13): [True: 28.4k, False: 111k]
  ------------------
 1854|  28.4k|                dsp->mc.w_avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1855|  28.4k|                              bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, jnt_weight
 1856|  28.4k|                              HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  28.4k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1857|  28.4k|                break;
 1858|  16.1k|            case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1858:13): [True: 16.1k, False: 123k]
  ------------------
 1859|  43.0k|            case COMP_INTER_SEG:
  ------------------
  |  Branch (1859:13): [True: 26.9k, False: 112k]
  ------------------
 1860|  43.0k|                dsp->mc.mask(uvdst, f->cur.stride[1],
 1861|  43.0k|                             tmp[b->mask_sign], tmp[!b->mask_sign],
 1862|  43.0k|                             bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, mask
 1863|  43.0k|                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  43.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1864|  43.0k|                break;
 1865|   139k|            }
 1866|   139k|        }
 1867|   100k|    }
 1868|       |
 1869|  1.13M|    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  1.13M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.13M]
  |  |  ------------------
  |  |   35|  1.13M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.13M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1870|      0|        hex_dump(dst, f->cur.stride[0], b_dim[0] * 4, b_dim[1] * 4, "y-pred");
 1871|      0|        if (has_chroma) {
  ------------------
  |  Branch (1871:13): [True: 0, False: 0]
  ------------------
 1872|      0|            hex_dump(&((pixel *) f->cur.data[1])[uvdstoff], f->cur.stride[1],
 1873|      0|                     cbw4 * 4, cbh4 * 4, "u-pred");
 1874|      0|            hex_dump(&((pixel *) f->cur.data[2])[uvdstoff], f->cur.stride[1],
 1875|      0|                     cbw4 * 4, cbh4 * 4, "v-pred");
 1876|      0|        }
 1877|      0|    }
 1878|       |
 1879|  1.13M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1880|       |
 1881|  1.13M|    if (b->skip) {
  ------------------
  |  Branch (1881:9): [True: 899k, False: 235k]
  ------------------
 1882|       |        // reset coef contexts
 1883|   899k|        BlockContext *const a = t->a;
 1884|   899k|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
 1885|   899k|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
 1886|   899k|        if (has_chroma) {
  ------------------
  |  Branch (1886:13): [True: 238k, False: 661k]
  ------------------
 1887|   238k|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
 1888|   238k|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
 1889|   238k|            memset_cw(&a->ccoef[0][cbx4], 0x40);
 1890|   238k|            memset_cw(&a->ccoef[1][cbx4], 0x40);
 1891|   238k|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
 1892|   238k|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
 1893|   238k|        }
 1894|   899k|        return 0;
 1895|   899k|    }
 1896|       |
 1897|   235k|    const TxfmInfo *const uvtx = &dav1d_txfm_dimensions[b->uvtx];
 1898|   235k|    const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
 1899|   235k|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1900|       |
 1901|   472k|    for (int init_y = 0; init_y < bh4; init_y += 16) {
  ------------------
  |  Branch (1901:26): [True: 237k, False: 235k]
  ------------------
 1902|   481k|        for (int init_x = 0; init_x < bw4; init_x += 16) {
  ------------------
  |  Branch (1902:30): [True: 243k, False: 237k]
  ------------------
 1903|       |            // coefficient coding & inverse transforms
 1904|   243k|            int y_off = !!init_y, y;
 1905|   243k|            dst += PXSTRIDE(f->cur.stride[0]) * 4 * init_y;
 1906|   511k|            for (y = init_y, t->by += init_y; y < imin(h4, init_y + 16);
  ------------------
  |  Branch (1906:47): [True: 267k, False: 243k]
  ------------------
 1907|   267k|                 y += ytx->h, y_off++)
 1908|   267k|            {
 1909|   267k|                int x, x_off = !!init_x;
 1910|   672k|                for (x = init_x, t->bx += init_x; x < imin(w4, init_x + 16);
  ------------------
  |  Branch (1910:51): [True: 404k, False: 267k]
  ------------------
 1911|   404k|                     x += ytx->w, x_off++)
 1912|   404k|                {
 1913|   404k|                    read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
 1914|   404k|                                   x_off, y_off, &dst[x * 4]);
 1915|   404k|                    t->bx += ytx->w;
 1916|   404k|                }
 1917|   267k|                dst += PXSTRIDE(f->cur.stride[0]) * 4 * ytx->h;
 1918|   267k|                t->bx -= x;
 1919|   267k|                t->by += ytx->h;
 1920|   267k|            }
 1921|   243k|            dst -= PXSTRIDE(f->cur.stride[0]) * 4 * y;
 1922|   243k|            t->by -= y;
 1923|       |
 1924|       |            // chroma coefs and inverse transform
 1925|   462k|            if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1925:17): [True: 154k, False: 89.1k]
  |  Branch (1925:46): [True: 308k, False: 154k]
  ------------------
 1926|   308k|                pixel *uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff +
 1927|   308k|                    (PXSTRIDE(f->cur.stride[1]) * init_y * 4 >> ss_ver);
 1928|   308k|                for (y = init_y >> ss_ver, t->by += init_y;
 1929|   649k|                     y < imin(ch4, (init_y + 16) >> ss_ver); y += uvtx->h)
  ------------------
  |  Branch (1929:22): [True: 340k, False: 308k]
  ------------------
 1930|   340k|                {
 1931|   340k|                    int x;
 1932|   340k|                    for (x = init_x >> ss_hor, t->bx += init_x;
 1933|   808k|                         x < imin(cw4, (init_x + 16) >> ss_hor); x += uvtx->w)
  ------------------
  |  Branch (1933:26): [True: 467k, False: 340k]
  ------------------
 1934|   467k|                    {
 1935|   467k|                        coef *cf;
 1936|   467k|                        int eob;
 1937|   467k|                        enum TxfmType txtp;
 1938|   467k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1938:29): [True: 0, False: 467k]
  ------------------
 1939|      0|                            const int p = t->frame_thread.pass & 1;
 1940|      0|                            const int cbi = *ts->frame_thread[p].cbi++;
 1941|      0|                            cf = ts->frame_thread[p].cf;
 1942|      0|                            ts->frame_thread[p].cf += uvtx->w * uvtx->h * 16;
 1943|      0|                            eob  = cbi >> 5;
 1944|      0|                            txtp = cbi & 0x1f;
 1945|   467k|                        } else {
 1946|   467k|                            uint8_t cf_ctx;
 1947|   467k|                            cf = bitfn(t->cf);
  ------------------
  |  |   77|   467k|#define bitfn(x) x##_16bpc
  ------------------
 1948|   467k|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
 1949|   467k|                                                        bx4 + (x << ss_hor)];
 1950|   467k|                            eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1951|   467k|                                               &t->l.ccoef[pl][cby4 + y],
 1952|   467k|                                               b->uvtx, bs, b, 0, 1 + pl,
 1953|   467k|                                               cf, &txtp, &cf_ctx);
 1954|   467k|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   467k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 467k]
  |  |  ------------------
  |  |   35|   467k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   467k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1955|      0|                                printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1956|      0|                                       "txtp=%d,eob=%d]: r=%d\n",
 1957|      0|                                       pl, b->uvtx, txtp, eob, ts->msac.rng);
 1958|   467k|                            int ctw = imin(uvtx->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1959|   467k|                            int cth = imin(uvtx->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1960|   467k|                            dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1961|   467k|                            dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1962|   467k|                        }
 1963|   467k|                        if (eob >= 0) {
  ------------------
  |  Branch (1963:29): [True: 142k, False: 325k]
  ------------------
 1964|   142k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   142k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 142k]
  |  |  ------------------
  |  |   35|   142k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   142k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1965|      0|                                coef_dump(cf, uvtx->h * 4, uvtx->w * 4, 3, "dq");
 1966|   142k|                            dsp->itx.itxfm_add[b->uvtx]
 1967|   142k|                                              [txtp](&uvdst[4 * x],
 1968|   142k|                                                     f->cur.stride[1],
 1969|   142k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   142k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1970|   142k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   142k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 142k]
  |  |  ------------------
  |  |   35|   142k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   142k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1971|      0|                                hex_dump(&uvdst[4 * x], f->cur.stride[1],
 1972|      0|                                         uvtx->w * 4, uvtx->h * 4, "recon");
 1973|   142k|                        }
 1974|   467k|                        t->bx += uvtx->w << ss_hor;
 1975|   467k|                    }
 1976|   340k|                    uvdst += PXSTRIDE(f->cur.stride[1]) * 4 * uvtx->h;
 1977|   340k|                    t->bx -= x << ss_hor;
 1978|   340k|                    t->by += uvtx->h << ss_ver;
 1979|   340k|                }
 1980|   308k|                t->by -= y << ss_ver;
 1981|   308k|            }
 1982|   243k|        }
 1983|   237k|    }
 1984|   235k|    return 0;
 1985|  1.13M|}
dav1d_filter_sbrow_deblock_cols_16bpc:
 1987|  78.5k|void bytefn(dav1d_filter_sbrow_deblock_cols)(Dav1dFrameContext *const f, const int sby) {
 1988|  78.5k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK) ||
  ------------------
  |  Branch (1988:9): [True: 0, False: 78.5k]
  ------------------
 1989|  78.5k|        (!f->frame_hdr->loopfilter.level_y[0] && !f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (1989:10): [True: 59.5k, False: 18.9k]
  |  Branch (1989:50): [True: 46.7k, False: 12.8k]
  ------------------
 1990|  46.7k|    {
 1991|  46.7k|        return;
 1992|  46.7k|    }
 1993|  31.7k|    const int y = sby * f->sb_step * 4;
 1994|  31.7k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1995|  31.7k|    pixel *const p[3] = {
 1996|  31.7k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 1997|  31.7k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 1998|  31.7k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 1999|  31.7k|    };
 2000|  31.7k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2001|  31.7k|    bytefn(dav1d_loopfilter_sbrow_cols)(f, p, mask, sby,
  ------------------
  |  |   87|  31.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  31.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2002|  31.7k|                                        f->lf.start_of_tile_row[sby]);
 2003|  31.7k|}
dav1d_filter_sbrow_deblock_rows_16bpc:
 2005|  78.5k|void bytefn(dav1d_filter_sbrow_deblock_rows)(Dav1dFrameContext *const f, const int sby) {
 2006|  78.5k|    const int y = sby * f->sb_step * 4;
 2007|  78.5k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2008|  78.5k|    pixel *const p[3] = {
 2009|  78.5k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2010|  78.5k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2011|  78.5k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2012|  78.5k|    };
 2013|  78.5k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2014|  78.5k|    if (f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK &&
  ------------------
  |  Branch (2014:9): [True: 78.5k, False: 0]
  ------------------
 2015|  78.5k|        (f->frame_hdr->loopfilter.level_y[0] || f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (2015:10): [True: 18.9k, False: 59.5k]
  |  Branch (2015:49): [True: 12.8k, False: 46.7k]
  ------------------
 2016|  31.7k|    {
 2017|  31.7k|        bytefn(dav1d_loopfilter_sbrow_rows)(f, p, mask, sby);
  ------------------
  |  |   87|  31.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  31.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2018|  31.7k|    }
 2019|  78.5k|    if (f->seq_hdr->cdef || f->lf.restore_planes) {
  ------------------
  |  Branch (2019:9): [True: 24.9k, False: 53.5k]
  |  Branch (2019:29): [True: 6.37k, False: 47.1k]
  ------------------
 2020|       |        // Store loop filtered pixels required by CDEF / LR
 2021|  31.3k|        bytefn(dav1d_copy_lpf)(f, p, sby);
  ------------------
  |  |   87|  31.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  31.3k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2022|  31.3k|    }
 2023|  78.5k|}
dav1d_filter_sbrow_cdef_16bpc:
 2025|  24.9k|void bytefn(dav1d_filter_sbrow_cdef)(Dav1dTaskContext *const tc, const int sby) {
 2026|  24.9k|    const Dav1dFrameContext *const f = tc->f;
 2027|  24.9k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_CDEF)) return;
  ------------------
  |  Branch (2027:9): [True: 0, False: 24.9k]
  ------------------
 2028|  24.9k|    const int sbsz = f->sb_step;
 2029|  24.9k|    const int y = sby * sbsz * 4;
 2030|  24.9k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2031|  24.9k|    pixel *const p[3] = {
 2032|  24.9k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2033|  24.9k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2034|  24.9k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2035|  24.9k|    };
 2036|  24.9k|    Av1Filter *prev_mask = f->lf.mask + ((sby - 1) >> !f->seq_hdr->sb128) * f->sb128w;
 2037|  24.9k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2038|  24.9k|    const int start = sby * sbsz;
 2039|  24.9k|    if (sby) {
  ------------------
  |  Branch (2039:9): [True: 22.6k, False: 2.32k]
  ------------------
 2040|  22.6k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2041|  22.6k|        pixel *p_up[3] = {
 2042|  22.6k|            p[0] - 8 * PXSTRIDE(f->cur.stride[0]),
 2043|  22.6k|            p[1] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2044|  22.6k|            p[2] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2045|  22.6k|        };
 2046|  22.6k|        bytefn(dav1d_cdef_brow)(tc, p_up, prev_mask, start - 2, start, 1, sby);
  ------------------
  |  |   87|  22.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  22.6k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2047|  22.6k|    }
 2048|  24.9k|    const int n_blks = sbsz - 2 * (sby + 1 < f->sbh);
 2049|  24.9k|    const int end = imin(start + n_blks, f->bh);
 2050|  24.9k|    bytefn(dav1d_cdef_brow)(tc, p, mask, start, end, 0, sby);
  ------------------
  |  |   87|  24.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  24.9k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2051|  24.9k|}
dav1d_filter_sbrow_resize_16bpc:
 2053|  5.83k|void bytefn(dav1d_filter_sbrow_resize)(Dav1dFrameContext *const f, const int sby) {
 2054|  5.83k|    const int sbsz = f->sb_step;
 2055|  5.83k|    const int y = sby * sbsz * 4;
 2056|  5.83k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2057|  5.83k|    const pixel *const p[3] = {
 2058|  5.83k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2059|  5.83k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2060|  5.83k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2061|  5.83k|    };
 2062|  5.83k|    pixel *const sr_p[3] = {
 2063|  5.83k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
 2064|  5.83k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
 2065|  5.83k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
 2066|  5.83k|    };
 2067|  5.83k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 2068|  20.7k|    for (int pl = 0; pl < 1 + 2 * has_chroma; pl++) {
  ------------------
  |  Branch (2068:22): [True: 14.8k, False: 5.83k]
  ------------------
 2069|  14.8k|        const int ss_ver = pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2069:28): [True: 9.05k, False: 5.83k]
  |  Branch (2069:34): [True: 5.83k, False: 3.22k]
  ------------------
 2070|  14.8k|        const int h_start = 8 * !!sby >> ss_ver;
 2071|  14.8k|        const ptrdiff_t dst_stride = f->sr_cur.p.stride[!!pl];
 2072|  14.8k|        pixel *dst = sr_p[pl] - h_start * PXSTRIDE(dst_stride);
 2073|  14.8k|        const ptrdiff_t src_stride = f->cur.stride[!!pl];
 2074|  14.8k|        const pixel *src = p[pl] - h_start * PXSTRIDE(src_stride);
 2075|  14.8k|        const int h_end = 4 * (sbsz - 2 * (sby + 1 < f->sbh)) >> ss_ver;
 2076|  14.8k|        const int ss_hor = pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2076:28): [True: 9.05k, False: 5.83k]
  |  Branch (2076:34): [True: 7.13k, False: 1.92k]
  ------------------
 2077|  14.8k|        const int dst_w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2078|  14.8k|        const int src_w = (4 * f->bw + ss_hor) >> ss_hor;
 2079|  14.8k|        const int img_h = (f->cur.p.h - sbsz * 4 * sby + ss_ver) >> ss_ver;
 2080|       |
 2081|  14.8k|        f->dsp->mc.resize(dst, dst_stride, src, src_stride, dst_w,
 2082|  14.8k|                          imin(img_h, h_end) + h_start, src_w,
 2083|  14.8k|                          f->resize_step[!!pl], f->resize_start[!!pl]
 2084|  14.8k|                          HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  14.8k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 2085|  14.8k|    }
 2086|  5.83k|}
dav1d_filter_sbrow_lr_16bpc:
 2088|  15.1k|void bytefn(dav1d_filter_sbrow_lr)(Dav1dFrameContext *const f, const int sby) {
 2089|  15.1k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_RESTORATION)) return;
  ------------------
  |  Branch (2089:9): [True: 0, False: 15.1k]
  ------------------
 2090|  15.1k|    const int y = sby * f->sb_step * 4;
 2091|  15.1k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2092|  15.1k|    pixel *const sr_p[3] = {
 2093|  15.1k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
 2094|  15.1k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
 2095|  15.1k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
 2096|  15.1k|    };
 2097|  15.1k|    bytefn(dav1d_lr_sbrow)(f, sr_p, sby);
  ------------------
  |  |   87|  15.1k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  15.1k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2098|  15.1k|}
dav1d_filter_sbrow_16bpc:
 2100|  78.5k|void bytefn(dav1d_filter_sbrow)(Dav1dFrameContext *const f, const int sby) {
 2101|  78.5k|    bytefn(dav1d_filter_sbrow_deblock_cols)(f, sby);
  ------------------
  |  |   87|  78.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  78.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2102|  78.5k|    bytefn(dav1d_filter_sbrow_deblock_rows)(f, sby);
  ------------------
  |  |   87|  78.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  78.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2103|  78.5k|    if (f->seq_hdr->cdef)
  ------------------
  |  Branch (2103:9): [True: 24.9k, False: 53.5k]
  ------------------
 2104|  24.9k|        bytefn(dav1d_filter_sbrow_cdef)(f->c->tc, sby);
  ------------------
  |  |   87|  24.9k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  24.9k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2105|  78.5k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1])
  ------------------
  |  Branch (2105:9): [True: 5.83k, False: 72.6k]
  ------------------
 2106|  5.83k|        bytefn(dav1d_filter_sbrow_resize)(f, sby);
  ------------------
  |  |   87|  5.83k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  5.83k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2107|  78.5k|    if (f->lf.restore_planes)
  ------------------
  |  Branch (2107:9): [True: 15.1k, False: 63.4k]
  ------------------
 2108|  15.1k|        bytefn(dav1d_filter_sbrow_lr)(f, sby);
  ------------------
  |  |   87|  15.1k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  15.1k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2109|  78.5k|}
dav1d_backup_ipred_edge_16bpc:
 2111|  88.9k|void bytefn(dav1d_backup_ipred_edge)(Dav1dTaskContext *const t) {
 2112|  88.9k|    const Dav1dFrameContext *const f = t->f;
 2113|  88.9k|    Dav1dTileState *const ts = t->ts;
 2114|  88.9k|    const int sby = t->by >> f->sb_shift;
 2115|  88.9k|    const int sby_off = f->sb128w * 128 * sby;
 2116|  88.9k|    const int x_off = ts->tiling.col_start;
 2117|       |
 2118|  88.9k|    const pixel *const y =
 2119|  88.9k|        ((const pixel *) f->cur.data[0]) + x_off * 4 +
 2120|  88.9k|                    ((t->by + f->sb_step) * 4 - 1) * PXSTRIDE(f->cur.stride[0]);
 2121|  88.9k|    pixel_copy(&f->ipred_edge[0][sby_off + x_off * 4], y,
  ------------------
  |  |   65|  88.9k|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
 2122|  88.9k|               4 * (ts->tiling.col_end - x_off));
 2123|       |
 2124|  88.9k|    if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (2124:9): [True: 56.5k, False: 32.4k]
  ------------------
 2125|  56.5k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2126|  56.5k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 2127|       |
 2128|  56.5k|        const ptrdiff_t uv_off = (x_off * 4 >> ss_hor) +
 2129|  56.5k|            (((t->by + f->sb_step) * 4 >> ss_ver) - 1) * PXSTRIDE(f->cur.stride[1]);
 2130|   169k|        for (int pl = 1; pl <= 2; pl++)
  ------------------
  |  Branch (2130:26): [True: 113k, False: 56.5k]
  ------------------
 2131|   113k|            pixel_copy(&f->ipred_edge[pl][sby_off + (x_off * 4 >> ss_hor)],
  ------------------
  |  |   65|   113k|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
 2132|  56.5k|                       &((const pixel *) f->cur.data[pl])[uv_off],
 2133|  56.5k|                       4 * (ts->tiling.col_end - x_off) >> ss_hor);
 2134|  56.5k|    }
 2135|  88.9k|}
dav1d_copy_pal_block_y_16bpc:
 2141|  58.5k|{
 2142|  58.5k|    const Dav1dFrameContext *const f = t->f;
 2143|  58.5k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2143:24): [True: 0, False: 58.5k]
  ------------------
 2144|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2145|      0|                            ((t->bx >> 1) + (t->by & 1))][0] :
 2146|  58.5k|        bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|  58.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  58.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2147|   313k|    for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2147:21): [True: 254k, False: 58.5k]
  ------------------
 2148|   254k|        memcpy(bytefn(t->al_pal)[0][bx4 + x][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   254k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   254k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2149|   222k|    for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2149:21): [True: 164k, False: 58.5k]
  ------------------
 2150|   164k|        memcpy(bytefn(t->al_pal)[1][by4 + y][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   164k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   164k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2151|  58.5k|}
dav1d_copy_pal_block_uv_16bpc:
 2157|  12.2k|{
 2158|  12.2k|    const Dav1dFrameContext *const f = t->f;
 2159|  12.2k|    const pixel (*const pal)[8] = t->frame_thread.pass ?
  ------------------
  |  Branch (2159:35): [True: 0, False: 12.2k]
  ------------------
 2160|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2161|      0|                            ((t->bx >> 1) + (t->by & 1))] :
 2162|  12.2k|        bytefn(t->scratch.pal);
  ------------------
  |  |   87|  12.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  12.2k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2163|       |    // see aomedia bug 2183 for why we use luma coordinates here
 2164|  36.6k|    for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (2164:22): [True: 24.4k, False: 12.2k]
  ------------------
 2165|   134k|        for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2165:25): [True: 110k, False: 24.4k]
  ------------------
 2166|   110k|            memcpy(bytefn(t->al_pal)[0][bx4 + x][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|   110k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   110k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2167|   105k|        for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2167:25): [True: 80.6k, False: 24.4k]
  ------------------
 2168|  80.6k|            memcpy(bytefn(t->al_pal)[1][by4 + y][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|  80.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  80.6k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2169|  24.4k|    }
 2170|  12.2k|}
dav1d_read_pal_plane_16bpc:
 2175|  70.7k|{
 2176|  70.7k|    Dav1dTileState *const ts = t->ts;
 2177|  70.7k|    const Dav1dFrameContext *const f = t->f;
 2178|  70.7k|    const int pal_sz = b->pal_sz[pl] = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  70.7k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 2179|  70.7k|                                           ts->cdf.m.pal_sz[pl][sz_ctx], 6) + 2;
 2180|  70.7k|    pixel cache[16], used_cache[8];
 2181|  70.7k|    int l_cache = pl ? t->pal_sz_uv[1][by4] : t->l.pal_sz[by4];
  ------------------
  |  Branch (2181:19): [True: 12.2k, False: 58.5k]
  ------------------
 2182|  70.7k|    int n_cache = 0;
 2183|       |    // don't reuse above palette outside SB64 boundaries
 2184|  70.7k|    int a_cache = by4 & 15 ? pl ? t->pal_sz_uv[0][bx4] : t->a->pal_sz[bx4] : 0;
  ------------------
  |  Branch (2184:19): [True: 55.0k, False: 15.7k]
  |  Branch (2184:30): [True: 10.1k, False: 44.9k]
  ------------------
 2185|  70.7k|    const pixel *l = bytefn(t->al_pal)[1][by4][pl];
  ------------------
  |  |   87|  70.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  70.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2186|  70.7k|    const pixel *a = bytefn(t->al_pal)[0][bx4][pl];
  ------------------
  |  |   87|  70.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  70.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2187|       |
 2188|       |    // fill/sort cache
 2189|   143k|    while (l_cache && a_cache) {
  ------------------
  |  Branch (2189:12): [True: 98.1k, False: 45.5k]
  |  Branch (2189:23): [True: 72.8k, False: 25.2k]
  ------------------
 2190|  72.8k|        if (*l < *a) {
  ------------------
  |  Branch (2190:13): [True: 33.4k, False: 39.4k]
  ------------------
 2191|  33.4k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2191:17): [True: 6.95k, False: 26.4k]
  |  Branch (2191:29): [True: 25.6k, False: 876]
  ------------------
 2192|  32.5k|                cache[n_cache++] = *l;
 2193|  33.4k|            l++;
 2194|  33.4k|            l_cache--;
 2195|  39.4k|        } else {
 2196|  39.4k|            if (*a == *l) {
  ------------------
  |  Branch (2196:17): [True: 13.6k, False: 25.7k]
  ------------------
 2197|  13.6k|                l++;
 2198|  13.6k|                l_cache--;
 2199|  13.6k|            }
 2200|  39.4k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2200:17): [True: 5.63k, False: 33.8k]
  |  Branch (2200:29): [True: 33.2k, False: 544]
  ------------------
 2201|  38.8k|                cache[n_cache++] = *a;
 2202|  39.4k|            a++;
 2203|  39.4k|            a_cache--;
 2204|  39.4k|        }
 2205|  72.8k|    }
 2206|  70.7k|    if (l_cache) {
  ------------------
  |  Branch (2206:9): [True: 25.2k, False: 45.5k]
  ------------------
 2207|   110k|        do {
 2208|   110k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2208:17): [True: 20.7k, False: 89.8k]
  |  Branch (2208:29): [True: 75.3k, False: 14.5k]
  ------------------
 2209|  96.0k|                cache[n_cache++] = *l;
 2210|   110k|            l++;
 2211|   110k|        } while (--l_cache > 0);
  ------------------
  |  Branch (2211:18): [True: 85.2k, False: 25.2k]
  ------------------
 2212|  45.5k|    } else if (a_cache) {
  ------------------
  |  Branch (2212:16): [True: 18.1k, False: 27.3k]
  ------------------
 2213|  81.3k|        do {
 2214|  81.3k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2214:17): [True: 12.2k, False: 69.1k]
  |  Branch (2214:29): [True: 58.1k, False: 10.9k]
  ------------------
 2215|  70.3k|                cache[n_cache++] = *a;
 2216|  81.3k|            a++;
 2217|  81.3k|        } while (--a_cache > 0);
  ------------------
  |  Branch (2217:18): [True: 63.1k, False: 18.1k]
  ------------------
 2218|  18.1k|    }
 2219|       |
 2220|       |    // find reused cache entries
 2221|  70.7k|    int i = 0;
 2222|   274k|    for (int n = 0; n < n_cache && i < pal_sz; n++)
  ------------------
  |  Branch (2222:21): [True: 214k, False: 59.8k]
  |  Branch (2222:36): [True: 203k, False: 10.8k]
  ------------------
 2223|   203k|        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|   203k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2223:13): [True: 92.2k, False: 110k]
  ------------------
 2224|  92.2k|            used_cache[i++] = cache[n];
 2225|  70.7k|    const int n_used_cache = i;
 2226|       |
 2227|       |    // parse new entries
 2228|  70.7k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2228:24): [True: 0, False: 70.7k]
  ------------------
 2229|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2230|      0|                            ((t->bx >> 1) + (t->by & 1))][pl] :
 2231|  70.7k|        bytefn(t->scratch.pal)[pl];
  ------------------
  |  |   87|  70.7k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  70.7k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2232|  70.7k|    if (i < pal_sz) {
  ------------------
  |  Branch (2232:9): [True: 57.3k, False: 13.3k]
  ------------------
 2233|  57.3k|        const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2233:25): [Folded, False: 57.3k]
  ------------------
 2234|  57.3k|        int prev = pal[i++] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2235|       |
 2236|  57.3k|        if (i < pal_sz) {
  ------------------
  |  Branch (2236:13): [True: 50.8k, False: 6.56k]
  ------------------
 2237|  50.8k|            int bits = bpc - 3 + dav1d_msac_decode_bools(&ts->msac, 2);
 2238|  50.8k|            const int max = (1 << bpc) - 1;
 2239|       |
 2240|   135k|            do {
 2241|   135k|                const int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2242|   135k|                prev = pal[i++] = imin(prev + delta + !pl, max);
 2243|   135k|                if (prev + !pl >= max) {
  ------------------
  |  Branch (2243:21): [True: 19.3k, False: 116k]
  ------------------
 2244|  54.0k|                    for (; i < pal_sz; i++)
  ------------------
  |  Branch (2244:28): [True: 34.7k, False: 19.3k]
  ------------------
 2245|  34.7k|                        pal[i] = max;
 2246|  19.3k|                    break;
 2247|  19.3k|                }
 2248|   116k|                bits = imin(bits, 1 + ulog2(max - prev - !pl));
 2249|   116k|            } while (i < pal_sz);
  ------------------
  |  Branch (2249:22): [True: 84.6k, False: 31.5k]
  ------------------
 2250|  50.8k|        }
 2251|       |
 2252|       |        // merge cache+new entries
 2253|  57.3k|        int n = 0, m = n_used_cache;
 2254|   341k|        for (i = 0; i < pal_sz; i++) {
  ------------------
  |  Branch (2254:21): [True: 284k, False: 57.3k]
  ------------------
 2255|   284k|            if (n < n_used_cache && (m >= pal_sz || used_cache[n] <= pal[m])) {
  ------------------
  |  Branch (2255:17): [True: 98.6k, False: 185k]
  |  Branch (2255:38): [True: 20.8k, False: 77.8k]
  |  Branch (2255:53): [True: 35.8k, False: 41.9k]
  ------------------
 2256|  56.7k|                pal[i] = used_cache[n++];
 2257|   227k|            } else {
 2258|   227k|                assert(m < pal_sz);
  ------------------
  |  Branch (2258:17): [True: 227k, False: 0]
  ------------------
 2259|   227k|                pal[i] = pal[m++];
 2260|   227k|            }
 2261|   284k|        }
 2262|  57.3k|    } else {
 2263|  13.3k|        memcpy(pal, used_cache, n_used_cache * sizeof(*used_cache));
 2264|  13.3k|    }
 2265|       |
 2266|  70.7k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  70.7k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 70.7k]
  |  |  ------------------
  |  |   35|  70.7k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  70.7k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2267|      0|        printf("Post-pal[pl=%d,sz=%d,cache_size=%d,used_cache=%d]: r=%d, cache=",
 2268|      0|               pl, pal_sz, n_cache, n_used_cache, ts->msac.rng);
 2269|      0|        for (int n = 0; n < n_cache; n++)
  ------------------
  |  Branch (2269:25): [True: 0, False: 0]
  ------------------
 2270|      0|            printf("%c%02x", n ? ' ' : '[', cache[n]);
  ------------------
  |  Branch (2270:30): [True: 0, False: 0]
  ------------------
 2271|      0|        printf("%s, pal=", n_cache ? "]" : "[]");
  ------------------
  |  Branch (2271:28): [True: 0, False: 0]
  ------------------
 2272|      0|        for (int n = 0; n < pal_sz; n++)
  ------------------
  |  Branch (2272:25): [True: 0, False: 0]
  ------------------
 2273|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2273:30): [True: 0, False: 0]
  ------------------
 2274|      0|        printf("]\n");
 2275|      0|    }
 2276|  70.7k|}
dav1d_read_pal_uv_16bpc:
 2280|  12.2k|{
 2281|  12.2k|    bytefn(dav1d_read_pal_plane)(t, b, 1, sz_ctx, bx4, by4);
  ------------------
  |  |   87|  12.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  12.2k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2282|       |
 2283|       |    // V pal coding
 2284|  12.2k|    Dav1dTileState *const ts = t->ts;
 2285|  12.2k|    const Dav1dFrameContext *const f = t->f;
 2286|  12.2k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2286:24): [True: 0, False: 12.2k]
  ------------------
 2287|      0|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2288|      0|                            ((t->bx >> 1) + (t->by & 1))][2] :
 2289|  12.2k|        bytefn(t->scratch.pal)[2];
  ------------------
  |  |   87|  12.2k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  12.2k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2290|  12.2k|    const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2290:21): [Folded, False: 12.2k]
  ------------------
 2291|  12.2k|    if (dav1d_msac_decode_bool_equi(&ts->msac)) {
  ------------------
  |  |   53|  12.2k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2291:9): [True: 6.19k, False: 6.02k]
  ------------------
 2292|  6.19k|        const int bits = bpc - 4 + dav1d_msac_decode_bools(&ts->msac, 2);
 2293|  6.19k|        int prev = pal[0] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2294|  6.19k|        const int max = (1 << bpc) - 1;
 2295|  28.4k|        for (int i = 1; i < b->pal_sz[1]; i++) {
  ------------------
  |  Branch (2295:25): [True: 22.2k, False: 6.19k]
  ------------------
 2296|  22.2k|            int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2297|  22.2k|            if (delta && dav1d_msac_decode_bool_equi(&ts->msac)) delta = -delta;
  ------------------
  |  |   53|  22.0k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2297:17): [True: 22.0k, False: 212]
  |  Branch (2297:26): [True: 10.7k, False: 11.3k]
  ------------------
 2298|  22.2k|            prev = pal[i] = (prev + delta) & max;
 2299|  22.2k|        }
 2300|  6.19k|    } else {
 2301|  30.6k|        for (int i = 0; i < b->pal_sz[1]; i++)
  ------------------
  |  Branch (2301:25): [True: 24.5k, False: 6.02k]
  ------------------
 2302|  24.5k|            pal[i] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2303|  6.02k|    }
 2304|  12.2k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  12.2k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 12.2k]
  |  |  ------------------
  |  |   35|  12.2k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  12.2k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2305|      0|        printf("Post-pal[pl=2]: r=%d ", ts->msac.rng);
 2306|      0|        for (int n = 0; n < b->pal_sz[1]; n++)
  ------------------
  |  Branch (2306:25): [True: 0, False: 0]
  ------------------
 2307|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2307:30): [True: 0, False: 0]
  ------------------
 2308|      0|        printf("]\n");
 2309|      0|    }
 2310|  12.2k|}

dav1d_ref_create:
   37|  80.0k|Dav1dRef *dav1d_ref_create(const enum AllocationType type, size_t size) {
   38|  80.0k|    size = (size + sizeof(void*) - 1) & ~(sizeof(void*) - 1);
   39|       |
   40|  80.0k|    uint8_t *const data = dav1d_alloc_aligned(type, size + sizeof(Dav1dRef), 64);
  ------------------
  |  |  134|  80.0k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
   41|  80.0k|    if (!data) return NULL;
  ------------------
  |  Branch (41:9): [True: 0, False: 80.0k]
  ------------------
   42|       |
   43|  80.0k|    Dav1dRef *const res = (Dav1dRef*)(data + size);
   44|  80.0k|    res->const_data = res->user_data = res->data = data;
   45|  80.0k|    atomic_init(&res->ref_cnt, 1);
   46|  80.0k|    res->free_ref = 0;
   47|  80.0k|    res->free_callback = default_free_callback;
   48|       |
   49|  80.0k|    return res;
   50|  80.0k|}
dav1d_ref_create_using_pool:
   56|   125k|Dav1dRef *dav1d_ref_create_using_pool(Dav1dMemPool *const pool, size_t size) {
   57|   125k|    void *const buf = dav1d_mem_pool_pop(pool, size);
   58|   125k|    if (!buf) return NULL;
  ------------------
  |  Branch (58:9): [True: 0, False: 125k]
  ------------------
   59|       |
   60|       |    /* Store Dav1dRef inside the Dav1dMemPoolBuffer alignment padding */
   61|   125k|    assert(sizeof(Dav1dMemPoolBuffer) + sizeof(Dav1dRef) <= 64);
  ------------------
  |  Branch (61:5): [True: 125k, Folded]
  ------------------
   62|   125k|    Dav1dRef *const res = &((Dav1dRef*)buf)[-1];
   63|   125k|    res->data = buf;
   64|   125k|    res->const_data = pool;
   65|   125k|    atomic_init(&res->ref_cnt, 1);
   66|   125k|    res->free_ref = 0;
   67|   125k|    res->free_callback = pool_free_callback;
   68|   125k|    res->user_data = buf;
   69|       |
   70|   125k|    return res;
   71|   125k|}
dav1d_ref_dec:
   73|  8.59M|void dav1d_ref_dec(Dav1dRef **const pref) {
   74|  8.59M|    assert(pref != NULL);
  ------------------
  |  Branch (74:5): [True: 8.59M, False: 0]
  ------------------
   75|       |
   76|  8.59M|    Dav1dRef *const ref = *pref;
   77|  8.59M|    if (!ref) return;
  ------------------
  |  Branch (77:9): [True: 6.23M, False: 2.36M]
  ------------------
   78|       |
   79|  2.36M|    *pref = NULL;
   80|  2.36M|    if (atomic_fetch_sub(&ref->ref_cnt, 1) == 1) {
  ------------------
  |  Branch (80:9): [True: 261k, False: 2.10M]
  ------------------
   81|   261k|        const int free_ref = ref->free_ref;
   82|   261k|        ref->free_callback(ref->const_data, ref->user_data);
   83|   261k|        if (free_ref) dav1d_free(ref);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  |  Branch (83:13): [True: 0, False: 261k]
  ------------------
   84|   261k|    }
   85|  2.36M|}
ref.c:default_free_callback:
   32|  80.0k|static void default_free_callback(const uint8_t *const data, void *const user_data) {
   33|  80.0k|    assert(data == user_data);
  ------------------
  |  Branch (33:5): [True: 80.0k, False: 0]
  ------------------
   34|  80.0k|    dav1d_free_aligned(user_data);
  ------------------
  |  |  136|  80.0k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
   35|  80.0k|}
ref.c:pool_free_callback:
   52|   125k|static void pool_free_callback(const uint8_t *const data, void *const user_data) {
   53|   125k|    dav1d_mem_pool_push((Dav1dMemPool*)data, user_data);
   54|   125k|}

obu.c:dav1d_ref_is_writable:
   73|  76.0k|static inline int dav1d_ref_is_writable(Dav1dRef *const ref) {
   74|  76.0k|    return atomic_load(&ref->ref_cnt) == 1 && ref->data;
  ------------------
  |  Branch (74:12): [True: 76.0k, False: 0]
  |  Branch (74:47): [True: 76.0k, False: 0]
  ------------------
   75|  76.0k|}
obu.c:dav1d_ref_init:
   59|    297|{
   60|    297|    ref->data = NULL;
   61|    297|    ref->const_data = ptr;
   62|       |    atomic_init(&ref->ref_cnt, 1);
   63|    297|    ref->free_ref = free_ref;
   64|    297|    ref->free_callback = free_callback;
   65|    297|    ref->user_data = user_data;
   66|    297|    return ref;
   67|    297|}
obu.c:dav1d_ref_inc:
   69|  2.54k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  2.54k|}
picture.c:dav1d_ref_inc:
   69|  1.67M|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  1.67M|}
picture.c:dav1d_ref_init:
   59|  54.9k|{
   60|  54.9k|    ref->data = NULL;
   61|  54.9k|    ref->const_data = ptr;
   62|       |    atomic_init(&ref->ref_cnt, 1);
   63|  54.9k|    ref->free_ref = free_ref;
   64|  54.9k|    ref->free_callback = free_callback;
   65|  54.9k|    ref->user_data = user_data;
   66|  54.9k|    return ref;
   67|  54.9k|}
cdf.c:dav1d_ref_inc:
   69|  94.1k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  94.1k|}
data.c:dav1d_ref_inc:
   69|   128k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|   128k|}
decode.c:dav1d_ref_inc:
   69|   198k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|   198k|}

dav1d_refmvs_find:
  354|  1.73M|{
  355|  1.73M|    const refmvs_frame *const rf = rt->rf;
  356|  1.73M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  357|  1.73M|    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
  358|  1.73M|    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
  359|  1.73M|    mv gmv[2], tgmv[2];
  360|       |
  361|  1.73M|    *cnt = 0;
  362|  1.73M|    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
  ------------------
  |  Branch (362:5): [True: 1.73M, False: 0]
  |  Branch (362:5): [True: 1.73M, False: 0]
  |  Branch (362:5): [True: 1.73M, False: 0]
  |  Branch (362:5): [True: 1.73M, False: 0]
  ------------------
  363|  1.73M|           ref.ref[1] >= -1 && ref.ref[1] <= 8);
  364|  1.73M|    if (ref.ref[0] > 0) {
  ------------------
  |  Branch (364:9): [True: 1.05M, False: 676k]
  ------------------
  365|  1.05M|        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
  366|  1.05M|                             bx4, by4, bw4, bh4, rf->frm_hdr);
  367|  1.05M|        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
  ------------------
  |  Branch (367:18): [True: 198k, False: 860k]
  ------------------
  368|   860k|                 tgmv[0] : (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|   860k|#define INVALID_MV 0x80008000
  ------------------
  369|  1.05M|    } else {
  370|   676k|        tgmv[0] = (mv) { .n = 0 };
  371|   676k|        gmv[0] = (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|   676k|#define INVALID_MV 0x80008000
  ------------------
  372|   676k|    }
  373|  1.73M|    if (ref.ref[1] > 0) {
  ------------------
  |  Branch (373:9): [True: 168k, False: 1.56M]
  ------------------
  374|   168k|        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
  375|   168k|                             bx4, by4, bw4, bh4, rf->frm_hdr);
  376|   168k|        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
  ------------------
  |  Branch (376:18): [True: 26.0k, False: 142k]
  ------------------
  377|   142k|                 tgmv[1] : (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|   142k|#define INVALID_MV 0x80008000
  ------------------
  378|   168k|    }
  379|       |
  380|       |    // top
  381|  1.73M|    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
  382|  1.73M|    unsigned max_rows = 0, n_rows = ~0;
  383|  1.73M|    const refmvs_block *b_top;
  384|  1.73M|    if (by4 > rt->tile_row.start) {
  ------------------
  |  Branch (384:9): [True: 1.22M, False: 510k]
  ------------------
  385|  1.22M|        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
  386|  1.22M|        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
  387|  1.22M|        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
  388|  1.22M|                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
  ------------------
  |  Branch (388:46): [True: 102k, False: 1.12M]
  ------------------
  389|  1.22M|                          &have_newmv, &have_row_mvs);
  390|  1.22M|    }
  391|       |
  392|       |    // left
  393|  1.73M|    unsigned max_cols = 0, n_cols = ~0U;
  394|  1.73M|    refmvs_block *const *b_left;
  395|  1.73M|    if (bx4 > rt->tile_col.start) {
  ------------------
  |  Branch (395:9): [True: 1.66M, False: 75.9k]
  ------------------
  396|  1.66M|        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
  397|  1.66M|        b_left = &rt->r[(by4 & 31) + 5];
  398|  1.66M|        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
  399|  1.66M|                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
  ------------------
  |  Branch (399:55): [True: 193k, False: 1.46M]
  ------------------
  400|  1.66M|                          &have_newmv, &have_col_mvs);
  401|  1.66M|    }
  402|       |
  403|       |    // top/right
  404|  1.73M|    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
  ------------------
  |  Branch (404:9): [True: 1.22M, False: 510k]
  |  Branch (404:26): [True: 722k, False: 502k]
  ------------------
  405|   722k|        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
  ------------------
  |  Branch (405:9): [True: 689k, False: 33.4k]
  |  Branch (405:33): [True: 645k, False: 44.0k]
  ------------------
  406|   645k|    {
  407|   645k|        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
  408|   645k|                              &have_newmv, &have_row_mvs);
  409|   645k|    }
  410|       |
  411|  1.73M|    const int nearest_match = have_col_mvs + have_row_mvs;
  412|  1.73M|    const int nearest_cnt = *cnt;
  413|  3.90M|    for (int n = 0; n < nearest_cnt; n++)
  ------------------
  |  Branch (413:21): [True: 2.17M, False: 1.73M]
  ------------------
  414|  2.17M|        mvstack[n].weight += 640;
  415|       |
  416|       |    // temporal
  417|  1.73M|    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
  418|  1.73M|    if (rf->use_ref_frame_mvs) {
  ------------------
  |  Branch (418:9): [True: 207k, False: 1.52M]
  ------------------
  419|   207k|        const ptrdiff_t stride = rf->rp_stride;
  420|   207k|        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
  421|   207k|        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
  422|   207k|        const refmvs_temporal_block *rb = rbi;
  423|   207k|        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
  ------------------
  |  Branch (423:28): [True: 12.2k, False: 195k]
  |  Branch (423:56): [True: 14.6k, False: 193k]
  ------------------
  424|   207k|        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
  425|   569k|        for (int y = 0; y < h8; y += step_v) {
  ------------------
  |  Branch (425:25): [True: 362k, False: 207k]
  ------------------
  426|  1.12M|            for (int x = 0; x < w8; x+= step_h) {
  ------------------
  |  Branch (426:29): [True: 762k, False: 362k]
  ------------------
  427|   762k|                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
  428|   762k|                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
  ------------------
  |  Branch (428:40): [True: 207k, False: 554k]
  ------------------
  429|   762k|            }
  430|   362k|            rb += stride * step_v;
  431|   362k|        }
  432|   207k|        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
  ------------------
  |  Branch (432:13): [True: 142k, False: 65.4k]
  |  Branch (432:36): [True: 126k, False: 16.0k]
  ------------------
  433|   126k|            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
  434|   126k|            rb = &rbi[bh8 * stride];
  435|   126k|            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
  436|   126k|                                                    (by8 & ~7) + 8);
  437|   126k|            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
  ------------------
  |  Branch (437:17): [True: 88.9k, False: 37.2k]
  |  Branch (437:31): [True: 68.1k, False: 20.7k]
  ------------------
  438|  68.1k|                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
  439|  68.1k|                                       NULL, NULL);
  440|  68.1k|            }
  441|   126k|            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
  ------------------
  |  Branch (441:17): [True: 93.0k, False: 33.1k]
  ------------------
  442|  93.0k|                if (has_bottom) {
  ------------------
  |  Branch (442:21): [True: 67.5k, False: 25.4k]
  ------------------
  443|  67.5k|                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
  444|  67.5k|                                           NULL, NULL);
  445|  67.5k|                }
  446|  93.0k|                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
  ------------------
  |  Branch (446:21): [True: 92.3k, False: 683]
  ------------------
  447|  92.3k|                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
  448|  92.3k|                                           ref, NULL, NULL);
  449|  92.3k|                }
  450|  93.0k|            }
  451|   126k|        }
  452|   207k|    }
  453|  1.73M|    assert(*cnt <= 8);
  ------------------
  |  Branch (453:5): [True: 1.73M, False: 0]
  ------------------
  454|       |
  455|       |    // top/left (which, confusingly, is part of "secondary" references)
  456|  1.73M|    int have_dummy_newmv_match;
  457|  1.73M|    if ((n_rows | n_cols) != ~0U) {
  ------------------
  |  Branch (457:9): [True: 1.16M, False: 573k]
  ------------------
  458|  1.16M|        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
  459|  1.16M|                              &have_dummy_newmv_match, &have_row_mvs);
  460|  1.16M|    }
  461|       |
  462|       |    // "secondary" (non-direct neighbour) top & left edges
  463|       |    // what is different about secondary is that everything is now in 8x8 resolution
  464|  5.20M|    for (int n = 2; n <= 3; n++) {
  ------------------
  |  Branch (464:21): [True: 3.47M, False: 1.73M]
  ------------------
  465|  3.47M|        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
  ------------------
  |  Branch (465:13): [True: 1.76M, False: 1.70M]
  |  Branch (465:38): [True: 1.11M, False: 645k]
  ------------------
  466|  1.11M|            n_rows += scan_row(mvstack, cnt, ref, gmv,
  467|  1.11M|                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
  468|  1.11M|                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
  ------------------
  |  Branch (468:59): [True: 19.2k, False: 1.10M]
  ------------------
  469|  1.11M|                               &have_dummy_newmv_match, &have_row_mvs);
  470|  1.11M|        }
  471|       |
  472|  3.47M|        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
  ------------------
  |  Branch (472:13): [True: 2.54M, False: 931k]
  |  Branch (472:38): [True: 1.89M, False: 646k]
  ------------------
  473|  1.89M|            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
  474|  1.89M|                               bh4, h4, (bx4 - n * 2 + 1) | 1,
  475|  1.89M|                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
  ------------------
  |  Branch (475:50): [True: 137k, False: 1.75M]
  ------------------
  476|  1.89M|                               &have_dummy_newmv_match, &have_col_mvs);
  477|  1.89M|        }
  478|  3.47M|    }
  479|  1.73M|    assert(*cnt <= 8);
  ------------------
  |  Branch (479:5): [True: 1.73M, False: 0]
  ------------------
  480|       |
  481|  1.73M|    const int ref_match_count = have_col_mvs + have_row_mvs;
  482|       |
  483|       |    // context build-up
  484|  1.73M|    int refmv_ctx, newmv_ctx;
  485|  1.73M|    switch (nearest_match) {
  ------------------
  |  Branch (485:13): [True: 1.73M, False: 0]
  ------------------
  486|   192k|    case 0:
  ------------------
  |  Branch (486:5): [True: 192k, False: 1.54M]
  ------------------
  487|   192k|        refmv_ctx = imin(2, ref_match_count);
  488|   192k|        newmv_ctx = ref_match_count > 0;
  489|   192k|        break;
  490|   707k|    case 1:
  ------------------
  |  Branch (490:5): [True: 707k, False: 1.02M]
  ------------------
  491|   707k|        refmv_ctx = imin(ref_match_count * 3, 4);
  492|   707k|        newmv_ctx = 3 - have_newmv;
  493|   707k|        break;
  494|   835k|    case 2:
  ------------------
  |  Branch (494:5): [True: 835k, False: 900k]
  ------------------
  495|   835k|        refmv_ctx = 5;
  496|   835k|        newmv_ctx = 5 - have_newmv;
  497|   835k|        break;
  498|  1.73M|    }
  499|       |
  500|       |    // sorting (nearest, then "secondary")
  501|  1.73M|    int len = nearest_cnt;
  502|  3.58M|    while (len) {
  ------------------
  |  Branch (502:12): [True: 1.84M, False: 1.73M]
  ------------------
  503|  1.84M|        int last = 0;
  504|  2.60M|        for (int n = 1; n < len; n++) {
  ------------------
  |  Branch (504:25): [True: 757k, False: 1.84M]
  ------------------
  505|   757k|            if (mvstack[n - 1].weight < mvstack[n].weight) {
  ------------------
  |  Branch (505:17): [True: 326k, False: 430k]
  ------------------
  506|   326k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  507|   326k|                EXCHANGE(mvstack[n - 1], mvstack[n]);
  ------------------
  |  |  506|   326k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (506:80): [Folded, False: 326k]
  |  |  ------------------
  ------------------
  508|   326k|                last = n;
  509|   326k|            }
  510|   757k|        }
  511|  1.84M|        len = last;
  512|  1.84M|    }
  513|  1.73M|    len = *cnt;
  514|  2.75M|    while (len > nearest_cnt) {
  ------------------
  |  Branch (514:12): [True: 1.01M, False: 1.73M]
  ------------------
  515|  1.01M|        int last = nearest_cnt;
  516|  1.66M|        for (int n = nearest_cnt + 1; n < len; n++) {
  ------------------
  |  Branch (516:39): [True: 641k, False: 1.01M]
  ------------------
  517|   641k|            if (mvstack[n - 1].weight < mvstack[n].weight) {
  ------------------
  |  Branch (517:17): [True: 157k, False: 484k]
  ------------------
  518|   157k|                EXCHANGE(mvstack[n - 1], mvstack[n]);
  ------------------
  |  |  506|   157k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (506:80): [Folded, False: 157k]
  |  |  ------------------
  ------------------
  519|   157k|#undef EXCHANGE
  520|   157k|                last = n;
  521|   157k|            }
  522|   641k|        }
  523|  1.01M|        len = last;
  524|  1.01M|    }
  525|       |
  526|  1.73M|    if (ref.ref[1] > 0) {
  ------------------
  |  Branch (526:9): [True: 168k, False: 1.56M]
  ------------------
  527|   168k|        if (*cnt < 2) {
  ------------------
  |  Branch (527:13): [True: 109k, False: 59.1k]
  ------------------
  528|   109k|            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
  529|   109k|            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
  530|   109k|            const int sz4 = imin(w4, h4);
  531|   109k|            refmvs_candidate *const same = &mvstack[*cnt];
  532|   109k|            int same_count[4] = { 0 };
  533|       |
  534|       |            // non-self references in top
  535|   191k|            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
  ------------------
  |  Branch (535:17): [True: 91.6k, False: 18.1k]
  |  Branch (535:48): [True: 100k, False: 91.6k]
  ------------------
  536|   100k|                const refmvs_block *const cand_b = &b_top[x];
  537|   100k|                add_compound_extended_candidate(same, same_count, cand_b,
  538|   100k|                                                sign0, sign1, ref, rf->sign_bias);
  539|   100k|                x += dav1d_block_dimensions[cand_b->bs][0];
  540|   100k|            }
  541|       |
  542|       |            // non-self references in left
  543|   221k|            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
  ------------------
  |  Branch (543:17): [True: 102k, False: 6.95k]
  |  Branch (543:48): [True: 118k, False: 102k]
  ------------------
  544|   118k|                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
  545|   118k|                add_compound_extended_candidate(same, same_count, cand_b,
  546|   118k|                                                sign0, sign1, ref, rf->sign_bias);
  547|   118k|                y += dav1d_block_dimensions[cand_b->bs][1];
  548|   118k|            }
  549|       |
  550|   109k|            refmvs_candidate *const diff = &same[2];
  551|   109k|            const int *const diff_count = &same_count[2];
  552|       |
  553|       |            // merge together
  554|   329k|            for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (554:29): [True: 219k, False: 109k]
  ------------------
  555|   219k|                int m = same_count[n];
  556|       |
  557|   219k|                if (m >= 2) continue;
  ------------------
  |  Branch (557:21): [True: 69.9k, False: 149k]
  ------------------
  558|       |
  559|   149k|                const int l = diff_count[n];
  560|   149k|                if (l) {
  ------------------
  |  Branch (560:21): [True: 136k, False: 13.2k]
  ------------------
  561|   136k|                    same[m].mv.mv[n] = diff[0].mv.mv[n];
  562|   136k|                    if (++m == 2) continue;
  ------------------
  |  Branch (562:25): [True: 90.5k, False: 45.8k]
  ------------------
  563|  45.8k|                    if (l == 2) {
  ------------------
  |  Branch (563:25): [True: 37.8k, False: 8.00k]
  ------------------
  564|  37.8k|                        same[1].mv.mv[n] = diff[1].mv.mv[n];
  565|  37.8k|                        continue;
  566|  37.8k|                    }
  567|  45.8k|                }
  568|  29.1k|                do {
  569|  29.1k|                    same[m].mv.mv[n] = tgmv[n];
  570|  29.1k|                } while (++m < 2);
  ------------------
  |  Branch (570:26): [True: 7.87k, False: 21.3k]
  ------------------
  571|  21.3k|            }
  572|       |
  573|       |            // if the first extended was the same as the non-extended one,
  574|       |            // then replace it with the second extended one
  575|   109k|            int n = *cnt;
  576|   109k|            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
  ------------------
  |  Branch (576:17): [True: 65.3k, False: 44.4k]
  |  Branch (576:27): [True: 46.9k, False: 18.4k]
  ------------------
  577|  46.9k|                mvstack[1].mv = mvstack[2].mv;
  578|   154k|            do {
  579|   154k|                mvstack[n].weight = 2;
  580|   154k|            } while (++n < 2);
  ------------------
  |  Branch (580:22): [True: 44.4k, False: 109k]
  ------------------
  581|   109k|            *cnt = 2;
  582|   109k|        }
  583|       |
  584|       |        // clamping
  585|   168k|        const int left = -(bx4 + bw4 + 4) * 4 * 8;
  586|   168k|        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
  587|   168k|        const int top = -(by4 + bh4 + 4) * 4 * 8;
  588|   168k|        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
  589|       |
  590|   168k|        const int n_refmvs = *cnt;
  591|   168k|        int n = 0;
  592|   384k|        do {
  593|   384k|            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
  594|   384k|            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
  595|   384k|            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
  596|   384k|            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
  597|   384k|        } while (++n < n_refmvs);
  ------------------
  |  Branch (597:18): [True: 215k, False: 168k]
  ------------------
  598|       |
  599|   168k|        switch (refmv_ctx >> 1) {
  ------------------
  |  Branch (599:17): [True: 168k, False: 0]
  ------------------
  600|  56.1k|        case 0:
  ------------------
  |  Branch (600:9): [True: 56.1k, False: 112k]
  ------------------
  601|  56.1k|            *ctx = imin(newmv_ctx, 1);
  602|  56.1k|            break;
  603|  61.2k|        case 1:
  ------------------
  |  Branch (603:9): [True: 61.2k, False: 107k]
  ------------------
  604|  61.2k|            *ctx = 1 + imin(newmv_ctx, 3);
  605|  61.2k|            break;
  606|  51.5k|        case 2:
  ------------------
  |  Branch (606:9): [True: 51.5k, False: 117k]
  ------------------
  607|  51.5k|            *ctx = iclip(3 + newmv_ctx, 4, 7);
  608|  51.5k|            break;
  609|   168k|        }
  610|       |
  611|   168k|        return;
  612|  1.56M|    } else if (*cnt < 2 && ref.ref[0] > 0) {
  ------------------
  |  Branch (612:16): [True: 654k, False: 912k]
  |  Branch (612:28): [True: 561k, False: 93.2k]
  ------------------
  613|   561k|        const int sign = rf->sign_bias[ref.ref[0] - 1];
  614|   561k|        const int sz4 = imin(w4, h4);
  615|       |
  616|       |        // non-self references in top
  617|  1.02M|        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
  ------------------
  |  Branch (617:13): [True: 489k, False: 72.3k]
  |  Branch (617:44): [True: 533k, False: 486k]
  |  Branch (617:55): [True: 531k, False: 2.37k]
  ------------------
  618|   531k|            const refmvs_block *const cand_b = &b_top[x];
  619|   531k|            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
  620|   531k|            x += dav1d_block_dimensions[cand_b->bs][0];
  621|   531k|        }
  622|       |
  623|       |        // non-self references in left
  624|   994k|        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
  ------------------
  |  Branch (624:13): [True: 501k, False: 60.3k]
  |  Branch (624:44): [True: 542k, False: 451k]
  |  Branch (624:55): [True: 493k, False: 49.1k]
  ------------------
  625|   493k|            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
  626|   493k|            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
  627|   493k|            y += dav1d_block_dimensions[cand_b->bs][1];
  628|   493k|        }
  629|   561k|    }
  630|  1.73M|    assert(*cnt <= 8);
  ------------------
  |  Branch (630:5): [True: 1.56M, False: 0]
  ------------------
  631|       |
  632|       |    // clamping
  633|  1.56M|    int n_refmvs = *cnt;
  634|  1.56M|    if (n_refmvs) {
  ------------------
  |  Branch (634:9): [True: 1.49M, False: 74.2k]
  ------------------
  635|  1.49M|        const int left = -(bx4 + bw4 + 4) * 4 * 8;
  636|  1.49M|        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
  637|  1.49M|        const int top = -(by4 + bh4 + 4) * 4 * 8;
  638|  1.49M|        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
  639|       |
  640|  1.49M|        int n = 0;
  641|  3.51M|        do {
  642|  3.51M|            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
  643|  3.51M|            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
  644|  3.51M|        } while (++n < n_refmvs);
  ------------------
  |  Branch (644:18): [True: 2.02M, False: 1.49M]
  ------------------
  645|  1.49M|    }
  646|       |
  647|  2.19M|    for (int n = *cnt; n < 2; n++)
  ------------------
  |  Branch (647:24): [True: 631k, False: 1.56M]
  ------------------
  648|   631k|        mvstack[n].mv.mv[0] = tgmv[0];
  649|       |
  650|  1.56M|    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
  651|  1.56M|}
dav1d_refmvs_tile_sbrow_init:
  657|   141k|{
  658|   141k|    if (rf->n_tile_threads == 1) tile_row_idx = 0;
  ------------------
  |  Branch (658:9): [True: 141k, False: 0]
  ------------------
  659|   141k|    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
  660|   141k|    const ptrdiff_t r_stride = rf->rp_stride * 2;
  661|   141k|    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
  ------------------
  |  Branch (661:33): [True: 0, False: 141k]
  |  Branch (661:60): [True: 0, False: 0]
  ------------------
  662|   141k|        35 * 2 * rf->n_blocks : 0;
  663|   141k|    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
  664|   141k|    const int sbsz = rf->sbsz;
  665|   141k|    const int off = (sbsz * sby) & 16;
  666|  3.87M|    for (int i = 0; i < sbsz; i++, r += r_stride)
  ------------------
  |  Branch (666:21): [True: 3.73M, False: 141k]
  ------------------
  667|  3.73M|        rt->r[off + 5 + i] = r;
  668|   141k|    rt->r[off + 0] = r;
  669|   141k|    r += r_stride;
  670|   141k|    rt->r[off + 1] = NULL;
  671|   141k|    rt->r[off + 2] = r;
  672|   141k|    r += r_stride;
  673|   141k|    rt->r[off + 3] = NULL;
  674|   141k|    rt->r[off + 4] = r;
  675|   141k|    if (sby & 1) {
  ------------------
  |  Branch (675:9): [True: 55.8k, False: 85.3k]
  ------------------
  676|  55.8k|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  677|  55.8k|        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
  ------------------
  |  |  676|  55.8k|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 55.8k]
  |  |  ------------------
  ------------------
  678|  55.8k|        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
  ------------------
  |  |  676|  55.8k|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 55.8k]
  |  |  ------------------
  ------------------
  679|  55.8k|        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
  ------------------
  |  |  676|  55.8k|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 55.8k]
  |  |  ------------------
  ------------------
  680|  55.8k|#undef EXCHANGE
  681|  55.8k|    }
  682|       |
  683|   141k|    rt->rf = rf;
  684|   141k|    rt->tile_row.start = tile_row_start4;
  685|   141k|    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
  686|   141k|    rt->tile_col.start = tile_col_start4;
  687|   141k|    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
  688|   141k|}
dav1d_refmvs_init_frame:
  812|  31.3k|{
  813|  31.3k|    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
  814|  31.3k|    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
  ------------------
  |  Branch (814:29): [True: 0, False: 31.3k]
  ------------------
  815|  31.3k|    const int n_blocks = rp_stride * n_tile_rows;
  816|       |
  817|  31.3k|    rf->sbsz = 16 << seq_hdr->sb128;
  818|  31.3k|    rf->frm_hdr = frm_hdr;
  819|  31.3k|    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
  820|  31.3k|    rf->ih8 = (frm_hdr->height + 7) >> 3;
  821|  31.3k|    rf->iw4 = rf->iw8 << 1;
  822|  31.3k|    rf->ih4 = rf->ih8 << 1;
  823|  31.3k|    rf->rp = rp;
  824|  31.3k|    rf->rp_stride = rp_stride;
  825|  31.3k|    rf->n_tile_threads = n_tile_threads;
  826|  31.3k|    rf->n_frame_threads = n_frame_threads;
  827|       |
  828|  31.3k|    if (n_blocks != rf->n_blocks) {
  ------------------
  |  Branch (828:9): [True: 7.15k, False: 24.2k]
  ------------------
  829|  7.15k|        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
  830|  7.15k|        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
  831|       |        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
  832|       |         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
  833|  7.15k|        dav1d_free_aligned(rf->r);
  ------------------
  |  |  136|  7.15k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  834|  7.15k|        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
  ------------------
  |  |  134|  7.15k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  835|  7.15k|        if (!rf->r) {
  ------------------
  |  Branch (835:13): [True: 0, False: 7.15k]
  ------------------
  836|      0|            rf->n_blocks = 0;
  837|      0|            return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  838|      0|        }
  839|       |
  840|  7.15k|        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
  841|  7.15k|        rf->n_blocks = n_blocks;
  842|  7.15k|    }
  843|       |
  844|  31.3k|    const int poc = frm_hdr->frame_offset;
  845|   250k|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (845:21): [True: 219k, False: 31.3k]
  ------------------
  846|   219k|        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
  847|   219k|                                          ref_poc[i], poc);
  848|   219k|        rf->sign_bias[i] = poc_diff > 0;
  849|   219k|        rf->mfmv_sign[i] = poc_diff < 0;
  850|   219k|        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
  851|   219k|                                            poc, ref_poc[i]), -31, 31);
  852|   219k|    }
  853|       |
  854|       |    // temporal MV setup
  855|  31.3k|    rf->n_mfmvs = 0;
  856|  31.3k|    rf->rp_ref = rp_ref;
  857|  31.3k|    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
  ------------------
  |  Branch (857:9): [True: 9.45k, False: 21.9k]
  |  Branch (857:39): [True: 9.45k, False: 0]
  ------------------
  858|  9.45k|        int total = 2;
  859|  9.45k|        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
  ------------------
  |  Branch (859:13): [True: 7.05k, False: 2.40k]
  |  Branch (859:26): [True: 3.62k, False: 3.43k]
  ------------------
  860|  3.62k|            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
  861|  3.62k|            total = 3;
  862|  3.62k|        }
  863|  9.45k|        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
  ------------------
  |  Branch (863:13): [True: 3.58k, False: 5.86k]
  |  Branch (863:26): [True: 866, False: 2.71k]
  ------------------
  864|  3.58k|                                      frm_hdr->frame_offset) > 0)
  865|    866|        {
  866|    866|            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
  867|    866|        }
  868|  9.45k|        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
  ------------------
  |  Branch (868:13): [True: 6.59k, False: 2.86k]
  |  Branch (868:26): [True: 585, False: 6.00k]
  ------------------
  869|  6.59k|                                      frm_hdr->frame_offset) > 0)
  870|    585|        {
  871|    585|            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
  872|    585|        }
  873|  9.45k|        if (rf->n_mfmvs < total && rp_ref[6] &&
  ------------------
  |  Branch (873:13): [True: 9.13k, False: 322]
  |  Branch (873:36): [True: 6.00k, False: 3.12k]
  ------------------
  874|  6.00k|            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
  ------------------
  |  Branch (874:13): [True: 3.12k, False: 2.88k]
  ------------------
  875|  6.00k|                         frm_hdr->frame_offset) > 0)
  876|  3.12k|        {
  877|  3.12k|            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
  878|  3.12k|        }
  879|  9.45k|        if (rf->n_mfmvs < total && rp_ref[1])
  ------------------
  |  Branch (879:13): [True: 8.35k, False: 1.09k]
  |  Branch (879:36): [True: 4.59k, False: 3.76k]
  ------------------
  880|  4.59k|            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
  881|       |
  882|  22.2k|        for (int n = 0; n < rf->n_mfmvs; n++) {
  ------------------
  |  Branch (882:25): [True: 12.7k, False: 9.45k]
  ------------------
  883|  12.7k|            const int rpoc = ref_poc[rf->mfmv_ref[n]];
  884|  12.7k|            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
  885|  12.7k|                                           rpoc, frm_hdr->frame_offset);
  886|  12.7k|            if (abs(diff1) > 31) {
  ------------------
  |  Branch (886:17): [True: 177, False: 12.6k]
  ------------------
  887|    177|                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
  ------------------
  |  |   41|    177|#define INVALID_REF2CUR (-32)
  ------------------
  888|  12.6k|            } else {
  889|  12.6k|                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
  ------------------
  |  Branch (889:39): [True: 8.15k, False: 4.45k]
  ------------------
  890|   100k|                for (int m = 0; m < 7; m++) {
  ------------------
  |  Branch (890:33): [True: 88.2k, False: 12.6k]
  ------------------
  891|  88.2k|                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
  892|  88.2k|                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
  893|  88.2k|                                                   rpoc, rrpoc);
  894|       |                    // unsigned comparison also catches the < 0 case
  895|  88.2k|                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
  ------------------
  |  Branch (895:46): [True: 23.4k, False: 64.8k]
  ------------------
  896|  88.2k|                }
  897|  12.6k|            }
  898|  12.7k|        }
  899|  9.45k|    }
  900|  31.3k|    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
  901|       |
  902|  31.3k|    return 0;
  903|  31.3k|}
dav1d_refmvs_dsp_init:
  926|  10.2k|{
  927|  10.2k|    c->load_tmvs = load_tmvs_c;
  928|  10.2k|    c->save_tmvs = save_tmvs_c;
  929|  10.2k|    c->splat_mv = splat_mv_c;
  930|       |
  931|  10.2k|#if HAVE_ASM
  932|       |#if ARCH_AARCH64 || ARCH_ARM
  933|       |    refmvs_dsp_init_arm(c);
  934|       |#elif ARCH_LOONGARCH64
  935|       |    refmvs_dsp_init_loongarch(c);
  936|       |#elif ARCH_X86
  937|       |    refmvs_dsp_init_x86(c);
  938|  10.2k|#endif
  939|  10.2k|#endif
  940|  10.2k|}
refmvs.c:scan_row:
  102|  2.34M|{
  103|  2.34M|    const refmvs_block *cand_b = b;
  104|  2.34M|    const enum BlockSize first_cand_bs = cand_b->bs;
  105|  2.34M|    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
  106|  2.34M|    int cand_bw4 = first_cand_b_dim[0];
  107|  2.34M|    int len = imax(step, imin(bw4, cand_bw4));
  108|       |
  109|  2.34M|    if (bw4 <= cand_bw4) {
  ------------------
  |  Branch (109:9): [True: 2.04M, False: 302k]
  ------------------
  110|       |        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
  111|       |        // position of the first block has to be odd already, i.e. not just
  112|       |        // for row_offset=-3/-5
  113|       |        // FIXME why can this not be cand_bw4?
  114|  2.04M|        const int weight = bw4 == 1 ? 2 :
  ------------------
  |  Branch (114:28): [True: 683k, False: 1.35M]
  ------------------
  115|  2.04M|                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
  116|  2.04M|        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
  117|  2.04M|                              have_newmv_match, have_refmv_match);
  118|  2.04M|        return weight >> 1;
  119|  2.04M|    }
  120|       |
  121|   625k|    for (int x = 0;;) {
  122|       |        // FIXME if we overhang above, we could fill a bitmask so we don't have
  123|       |        // to repeat the add_spatial_candidate() for the next row, but just increase
  124|       |        // the weight here
  125|   625k|        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
  126|   625k|                              have_newmv_match, have_refmv_match);
  127|   625k|        x += len;
  128|   625k|        if (x >= w4) return 1;
  ------------------
  |  Branch (128:13): [True: 302k, False: 322k]
  ------------------
  129|   322k|        cand_b = &b[x];
  130|   322k|        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
  131|   322k|        assert(cand_bw4 < bw4);
  ------------------
  |  Branch (131:9): [True: 322k, False: 0]
  ------------------
  132|   322k|        len = imax(step, cand_bw4);
  133|   322k|    }
  134|   302k|}
refmvs.c:scan_col:
  141|  3.55M|{
  142|  3.55M|    const refmvs_block *cand_b = &b[0][bx4];
  143|  3.55M|    const enum BlockSize first_cand_bs = cand_b->bs;
  144|  3.55M|    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
  145|  3.55M|    int cand_bh4 = first_cand_b_dim[1];
  146|  3.55M|    int len = imax(step, imin(bh4, cand_bh4));
  147|       |
  148|  3.55M|    if (bh4 <= cand_bh4) {
  ------------------
  |  Branch (148:9): [True: 3.16M, False: 393k]
  ------------------
  149|       |        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
  150|       |        // position of the first block has to be odd already, i.e. not just
  151|       |        // for col_offset=-3/-5
  152|       |        // FIXME why can this not be cand_bh4?
  153|  3.16M|        const int weight = bh4 == 1 ? 2 :
  ------------------
  |  Branch (153:28): [True: 1.29M, False: 1.86M]
  ------------------
  154|  3.16M|                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
  155|  3.16M|        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
  156|  3.16M|                            have_newmv_match, have_refmv_match);
  157|  3.16M|        return weight >> 1;
  158|  3.16M|    }
  159|       |
  160|   775k|    for (int y = 0;;) {
  161|       |        // FIXME if we overhang above, we could fill a bitmask so we don't have
  162|       |        // to repeat the add_spatial_candidate() for the next row, but just increase
  163|       |        // the weight here
  164|   775k|        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
  165|   775k|                              have_newmv_match, have_refmv_match);
  166|   775k|        y += len;
  167|   775k|        if (y >= h4) return 1;
  ------------------
  |  Branch (167:13): [True: 393k, False: 382k]
  ------------------
  168|   382k|        cand_b = &b[y][bx4];
  169|   382k|        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
  170|   382k|        assert(cand_bh4 < bh4);
  ------------------
  |  Branch (170:9): [True: 382k, False: 0]
  ------------------
  171|   382k|        len = imax(step, cand_bh4);
  172|   382k|    }
  173|   393k|}
refmvs.c:add_spatial_candidate:
   46|  8.41M|{
   47|  8.41M|    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
  ------------------
  |  |   40|  8.41M|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (47:9): [True: 702k, False: 7.70M]
  ------------------
   48|       |
   49|  7.70M|    if (ref.ref[1] == -1) {
  ------------------
  |  Branch (49:9): [True: 6.77M, False: 936k]
  ------------------
   50|  8.29M|        for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (50:25): [True: 7.58M, False: 709k]
  ------------------
   51|  7.58M|            if (b->ref.ref[n] == ref.ref[0]) {
  ------------------
  |  Branch (51:17): [True: 6.06M, False: 1.51M]
  ------------------
   52|  6.06M|                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
  ------------------
  |  |   40|  1.71M|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (52:37): [True: 1.71M, False: 4.34M]
  |  Branch (52:52): [True: 213k, False: 1.50M]
  ------------------
   53|  5.84M|                                   gmv[0] : b->mv.mv[n];
   54|       |
   55|  6.06M|                *have_refmv_match = 1;
   56|  6.06M|                *have_newmv_match |= b->mf >> 1;
   57|       |
   58|  6.06M|                const int last = *cnt;
   59|  10.1M|                for (int m = 0; m < last; m++)
  ------------------
  |  Branch (59:33): [True: 6.90M, False: 3.28M]
  ------------------
   60|  6.90M|                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
  ------------------
  |  Branch (60:25): [True: 2.77M, False: 4.12M]
  ------------------
   61|  2.77M|                        mvstack[m].weight += weight;
   62|  2.77M|                        return;
   63|  2.77M|                    }
   64|       |
   65|  3.28M|                if (last < 8) {
  ------------------
  |  Branch (65:21): [True: 3.27M, False: 8.25k]
  ------------------
   66|  3.27M|                    mvstack[last].mv.mv[0] = cand_mv;
   67|  3.27M|                    mvstack[last].weight = weight;
   68|  3.27M|                    *cnt = last + 1;
   69|  3.27M|                }
   70|  3.28M|                return;
   71|  6.06M|            }
   72|  7.58M|        }
   73|  6.77M|    } else if (b->ref.pair == ref.pair) {
  ------------------
  |  Branch (73:16): [True: 352k, False: 583k]
  ------------------
   74|   352k|        const refmvs_mvpair cand_mv = { .mv = {
   75|   352k|            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
  ------------------
  |  |   40|  18.5k|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (75:20): [True: 18.5k, False: 334k]
  |  Branch (75:35): [True: 5.60k, False: 12.9k]
  ------------------
   76|   352k|            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
  ------------------
  |  |   40|  18.5k|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (76:20): [True: 18.5k, False: 334k]
  |  Branch (76:35): [True: 4.23k, False: 14.2k]
  ------------------
   77|   352k|        }};
   78|       |
   79|   352k|        *have_refmv_match = 1;
   80|   352k|        *have_newmv_match |= b->mf >> 1;
   81|       |
   82|   352k|        const int last = *cnt;
   83|   536k|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (83:25): [True: 329k, False: 207k]
  ------------------
   84|   329k|            if (mvstack[n].mv.n == cand_mv.n) {
  ------------------
  |  Branch (84:17): [True: 145k, False: 184k]
  ------------------
   85|   145k|                mvstack[n].weight += weight;
   86|   145k|                return;
   87|   145k|            }
   88|       |
   89|   207k|        if (last < 8) {
  ------------------
  |  Branch (89:13): [True: 207k, False: 605]
  ------------------
   90|   207k|            mvstack[last].mv = cand_mv;
   91|   207k|            mvstack[last].weight = weight;
   92|   207k|            *cnt = last + 1;
   93|   207k|        }
   94|   207k|    }
   95|  7.70M|}
refmvs.c:add_temporal_candidate:
  198|   990k|{
  199|   990k|    if (rb->mv.n == INVALID_MV) return;
  ------------------
  |  |   40|   990k|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (199:9): [True: 552k, False: 437k]
  ------------------
  200|       |
  201|   437k|    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
  202|   437k|    fix_mv_precision(rf->frm_hdr, &mv);
  203|       |
  204|   437k|    const int last = *cnt;
  205|   437k|    if (ref.ref[1] == -1) {
  ------------------
  |  Branch (205:9): [True: 304k, False: 132k]
  ------------------
  206|   304k|        if (globalmv_ctx)
  ------------------
  |  Branch (206:13): [True: 68.4k, False: 236k]
  ------------------
  207|  68.4k|            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
  208|       |
  209|   843k|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (209:25): [True: 738k, False: 104k]
  ------------------
  210|   738k|            if (mvstack[n].mv.mv[0].n == mv.n) {
  ------------------
  |  Branch (210:17): [True: 199k, False: 538k]
  ------------------
  211|   199k|                mvstack[n].weight += 2;
  212|   199k|                return;
  213|   199k|            }
  214|   104k|        if (last < 8) {
  ------------------
  |  Branch (214:13): [True: 104k, False: 581]
  ------------------
  215|   104k|            mvstack[last].mv.mv[0] = mv;
  216|   104k|            mvstack[last].weight = 2;
  217|   104k|            *cnt = last + 1;
  218|   104k|        }
  219|   132k|    } else {
  220|   132k|        refmvs_mvpair mvp = { .mv = {
  221|   132k|            [0] = mv,
  222|   132k|            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
  223|   132k|        }};
  224|   132k|        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
  225|       |
  226|   236k|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (226:25): [True: 213k, False: 23.1k]
  ------------------
  227|   213k|            if (mvstack[n].mv.n == mvp.n) {
  ------------------
  |  Branch (227:17): [True: 109k, False: 103k]
  ------------------
  228|   109k|                mvstack[n].weight += 2;
  229|   109k|                return;
  230|   109k|            }
  231|  23.1k|        if (last < 8) {
  ------------------
  |  Branch (231:13): [True: 22.5k, False: 600]
  ------------------
  232|  22.5k|            mvstack[last].mv = mvp;
  233|  22.5k|            mvstack[last].weight = 2;
  234|  22.5k|            *cnt = last + 1;
  235|  22.5k|        }
  236|  23.1k|    }
  237|   437k|}
refmvs.c:mv_projection:
  175|   570k|static inline union mv mv_projection(const union mv mv, const int num, const int den) {
  176|   570k|    static const uint16_t div_mult[32] = {
  177|   570k|           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
  178|   570k|        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
  179|   570k|        1024,   963,  910,  862,  819,  780,  744,  712,
  180|   570k|         682,   655,  630,  606,  585,  564,  546,  528
  181|   570k|    };
  182|   570k|    assert(den > 0 && den < 32);
  ------------------
  |  Branch (182:5): [True: 570k, False: 0]
  |  Branch (182:5): [True: 570k, False: 0]
  ------------------
  183|   570k|    assert(num > -32 && num < 32);
  ------------------
  |  Branch (183:5): [True: 570k, False: 0]
  |  Branch (183:5): [True: 570k, False: 0]
  ------------------
  184|   570k|    const int frac = num * div_mult[den];
  185|   570k|    const int y = mv.y * frac, x = mv.x * frac;
  186|       |    // Round and clip according to AV1 spec section 7.9.3
  187|   570k|    return (union mv) { // 0x3fff == (1 << 14) - 1
  188|   570k|        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
  189|   570k|        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
  190|   570k|    };
  191|   570k|}
refmvs.c:add_compound_extended_candidate:
  245|   218k|{
  246|   218k|    refmvs_candidate *const diff = &same[2];
  247|   218k|    int *const diff_count = &same_count[2];
  248|       |
  249|   566k|    for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (249:21): [True: 426k, False: 139k]
  ------------------
  250|   426k|        const int cand_ref = cand_b->ref.ref[n];
  251|       |
  252|   426k|        if (cand_ref <= 0) break;
  ------------------
  |  Branch (252:13): [True: 78.7k, False: 347k]
  ------------------
  253|       |
  254|   347k|        mv cand_mv = cand_b->mv.mv[n];
  255|   347k|        if (cand_ref == ref.ref[0]) {
  ------------------
  |  Branch (255:13): [True: 130k, False: 216k]
  ------------------
  256|   130k|            if (same_count[0] < 2)
  ------------------
  |  Branch (256:17): [True: 123k, False: 6.50k]
  ------------------
  257|   123k|                same[same_count[0]++].mv.mv[0] = cand_mv;
  258|   130k|            if (diff_count[1] < 2) {
  ------------------
  |  Branch (258:17): [True: 107k, False: 22.5k]
  ------------------
  259|   107k|                if (sign1 ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (259:21): [True: 4.38k, False: 103k]
  ------------------
  260|  4.38k|                    cand_mv.y = -cand_mv.y;
  261|  4.38k|                    cand_mv.x = -cand_mv.x;
  262|  4.38k|                }
  263|   107k|                diff[diff_count[1]++].mv.mv[1] = cand_mv;
  264|   107k|            }
  265|   216k|        } else if (cand_ref == ref.ref[1]) {
  ------------------
  |  Branch (265:20): [True: 115k, False: 101k]
  ------------------
  266|   115k|            if (same_count[1] < 2)
  ------------------
  |  Branch (266:17): [True: 111k, False: 3.38k]
  ------------------
  267|   111k|                same[same_count[1]++].mv.mv[1] = cand_mv;
  268|   115k|            if (diff_count[0] < 2) {
  ------------------
  |  Branch (268:17): [True: 93.5k, False: 21.7k]
  ------------------
  269|  93.5k|                if (sign0 ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (269:21): [True: 3.59k, False: 90.0k]
  ------------------
  270|  3.59k|                    cand_mv.y = -cand_mv.y;
  271|  3.59k|                    cand_mv.x = -cand_mv.x;
  272|  3.59k|                }
  273|  93.5k|                diff[diff_count[0]++].mv.mv[0] = cand_mv;
  274|  93.5k|            }
  275|   115k|        } else {
  276|   101k|            mv i_cand_mv = (union mv) {
  277|   101k|                .x = -cand_mv.x,
  278|   101k|                .y = -cand_mv.y
  279|   101k|            };
  280|       |
  281|   101k|            if (diff_count[0] < 2) {
  ------------------
  |  Branch (281:17): [True: 76.5k, False: 25.1k]
  ------------------
  282|  76.5k|                diff[diff_count[0]++].mv.mv[0] =
  283|  76.5k|                    sign0 ^ sign_bias[cand_ref - 1] ?
  ------------------
  |  Branch (283:21): [True: 2.59k, False: 73.9k]
  ------------------
  284|  73.9k|                    i_cand_mv : cand_mv;
  285|  76.5k|            }
  286|       |
  287|   101k|            if (diff_count[1] < 2) {
  ------------------
  |  Branch (287:17): [True: 71.3k, False: 30.3k]
  ------------------
  288|  71.3k|                diff[diff_count[1]++].mv.mv[1] =
  289|  71.3k|                    sign1 ^ sign_bias[cand_ref - 1] ?
  ------------------
  |  Branch (289:21): [True: 2.88k, False: 68.4k]
  ------------------
  290|  68.4k|                    i_cand_mv : cand_mv;
  291|  71.3k|            }
  292|   101k|        }
  293|   347k|    }
  294|   218k|}
refmvs.c:add_single_extended_candidate:
  299|  1.02M|{
  300|  2.03M|    for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (300:21): [True: 2.00M, False: 28.8k]
  ------------------
  301|  2.00M|        const int cand_ref = cand_b->ref.ref[n];
  302|       |
  303|  2.00M|        if (cand_ref <= 0) break;
  ------------------
  |  Branch (303:13): [True: 995k, False: 1.01M]
  ------------------
  304|       |        // we need to continue even if cand_ref == ref.ref[0], since
  305|       |        // the candidate could have been added as a globalmv variant,
  306|       |        // which changes the value
  307|       |        // FIXME if scan_{row,col}() returned a mask for the nearest
  308|       |        // edge, we could skip the appropriate ones here
  309|       |
  310|  1.01M|        mv cand_mv = cand_b->mv.mv[n];
  311|  1.01M|        if (sign ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (311:13): [True: 4.53k, False: 1.00M]
  ------------------
  312|  4.53k|            cand_mv.y = -cand_mv.y;
  313|  4.53k|            cand_mv.x = -cand_mv.x;
  314|  4.53k|        }
  315|       |
  316|  1.01M|        int m;
  317|  1.01M|        const int last = *cnt;
  318|  1.11M|        for (m = 0; m < last; m++)
  ------------------
  |  Branch (318:21): [True: 985k, False: 133k]
  ------------------
  319|   985k|            if (cand_mv.n == mvstack[m].mv.mv[0].n)
  ------------------
  |  Branch (319:17): [True: 878k, False: 107k]
  ------------------
  320|   878k|                break;
  321|  1.01M|        if (m == last) {
  ------------------
  |  Branch (321:13): [True: 133k, False: 878k]
  ------------------
  322|   133k|            mvstack[m].mv.mv[0] = cand_mv;
  323|   133k|            mvstack[m].weight = 2; // "minimal"
  324|   133k|            *cnt = last + 1;
  325|   133k|        }
  326|  1.01M|    }
  327|  1.02M|}

decode.c:dav1d_refmvs_save_tmvs:
  145|  48.7k|{
  146|  48.7k|    const refmvs_frame *const rf = rt->rf;
  147|       |
  148|  48.7k|    assert(row_start8 >= 0);
  ------------------
  |  Branch (148:5): [True: 48.7k, False: 0]
  ------------------
  149|  48.7k|    assert((unsigned) (row_end8 - row_start8) <= 16U);
  ------------------
  |  Branch (149:5): [True: 48.7k, False: 0]
  ------------------
  150|  48.7k|    row_end8 = imin(row_end8, rf->ih8);
  151|  48.7k|    col_end8 = imin(col_end8, rf->iw8);
  152|       |
  153|  48.7k|    const ptrdiff_t stride = rf->rp_stride;
  154|  48.7k|    const uint8_t *const ref_sign = rf->mfmv_sign;
  155|  48.7k|    refmvs_temporal_block *rp = &rf->rp[row_start8 * stride];
  156|       |
  157|  48.7k|    dsp->save_tmvs(rp, stride, rt->r + 6, ref_sign,
  158|  48.7k|                   col_end8, row_end8, col_start8, row_start8);
  159|  48.7k|}

dav1d_init_last_nonzero_col_from_eob_tables:
  350|  2.52k|COLD void dav1d_init_last_nonzero_col_from_eob_tables(void) {
  351|       |    static pthread_once_t initted = PTHREAD_ONCE_INIT;
  352|  2.52k|    pthread_once(&initted, init_internal);
  353|  2.52k|}
scan.c:init_internal:
  333|      1|static COLD void init_internal(void) {
  334|      1|    init_tbl(last_nonzero_col_from_eob_4x4,   scan_4x4,    4,  4);
  335|      1|    init_tbl(last_nonzero_col_from_eob_8x8,   scan_8x8,    8,  8);
  336|      1|    init_tbl(last_nonzero_col_from_eob_16x16, scan_16x16, 16, 16);
  337|      1|    init_tbl(last_nonzero_col_from_eob_32x32, scan_32x32, 32, 32);
  338|      1|    init_tbl(last_nonzero_col_from_eob_4x8,   scan_4x8,    4,  8);
  339|      1|    init_tbl(last_nonzero_col_from_eob_8x4,   scan_8x4,    8,  4);
  340|      1|    init_tbl(last_nonzero_col_from_eob_8x16,  scan_8x16,   8, 16);
  341|      1|    init_tbl(last_nonzero_col_from_eob_16x8,  scan_16x8,  16,  8);
  342|      1|    init_tbl(last_nonzero_col_from_eob_16x32, scan_16x32, 16, 32);
  343|      1|    init_tbl(last_nonzero_col_from_eob_32x16, scan_32x16, 32, 16);
  344|      1|    init_tbl(last_nonzero_col_from_eob_4x16,  scan_4x16,   4, 16);
  345|      1|    init_tbl(last_nonzero_col_from_eob_16x4,  scan_16x4,  16,  4);
  346|      1|    init_tbl(last_nonzero_col_from_eob_8x32,  scan_8x32,   8, 32);
  347|      1|    init_tbl(last_nonzero_col_from_eob_32x8,  scan_32x8,  32,  8);
  348|      1|}
scan.c:init_tbl:
  321|     14|{
  322|     14|    int max_col = 0;
  323|    218|    for (int y = 0, n = 0; y < h; y++) {
  ------------------
  |  Branch (323:28): [True: 204, False: 14]
  ------------------
  324|  3.54k|        for (int x = 0; x < w; x++, n++) {
  ------------------
  |  Branch (324:25): [True: 3.34k, False: 204]
  ------------------
  325|  3.34k|            const int rc = scan[n];
  326|  3.34k|            const int rcx = rc & (h - 1);
  327|  3.34k|            max_col = imax(max_col, rcx);
  328|  3.34k|            last_nonzero_col_from_eob[n] = max_col;
  329|  3.34k|        }
  330|    204|    }
  331|     14|}

dav1d_get_shear_params:
   80|  80.9k|int dav1d_get_shear_params(Dav1dWarpedMotionParams *const wm) {
   81|  80.9k|    const int32_t *const mat = wm->matrix;
   82|       |
   83|  80.9k|    if (mat[2] <= 0) return 1;
  ------------------
  |  Branch (83:9): [True: 0, False: 80.9k]
  ------------------
   84|       |
   85|  80.9k|    wm->u.p.alpha = iclip_wmp(mat[2] - 0x10000);
   86|  80.9k|    wm->u.p.beta = iclip_wmp(mat[3]);
   87|       |
   88|  80.9k|    int shift;
   89|  80.9k|    const int y = apply_sign(resolve_divisor_32(abs(mat[2]), &shift), mat[2]);
   90|  80.9k|    const int64_t v1 = ((int64_t) mat[4] * 0x10000) * y;
   91|  80.9k|    const int rnd = (1 << shift) >> 1;
   92|  80.9k|    wm->u.p.gamma = iclip_wmp(apply_sign64((int) ((llabs(v1) + rnd) >> shift), v1));
   93|  80.9k|    const int64_t v2 = ((int64_t) mat[3] * mat[4]) * y;
   94|  80.9k|    wm->u.p.delta = iclip_wmp(mat[5] -
   95|  80.9k|                          apply_sign64((int) ((llabs(v2) + rnd) >> shift), v2) -
   96|  80.9k|                          0x10000);
   97|       |
   98|  80.9k|    return (4 * abs(wm->u.p.alpha) + 7 * abs(wm->u.p.beta) >= 0x10000) ||
  ------------------
  |  Branch (98:12): [True: 3.10k, False: 77.8k]
  ------------------
   99|  77.8k|           (4 * abs(wm->u.p.gamma) + 4 * abs(wm->u.p.delta) >= 0x10000);
  ------------------
  |  Branch (99:12): [True: 1.02k, False: 76.8k]
  ------------------
  100|  80.9k|}
dav1d_find_affine_int:
  153|  79.8k|{
  154|  79.8k|    int32_t *const mat = wm->matrix;
  155|  79.8k|    int a[2][2] = { { 0, 0 }, { 0, 0 } };
  156|  79.8k|    int bx[2] = { 0, 0 };
  157|  79.8k|    int by[2] = { 0, 0 };
  158|  79.8k|    const int rsuy = 2 * bh4 - 1;
  159|  79.8k|    const int rsux = 2 * bw4 - 1;
  160|  79.8k|    const int suy = rsuy * 8;
  161|  79.8k|    const int sux = rsux * 8;
  162|  79.8k|    const int duy = suy + mv.y;
  163|  79.8k|    const int dux = sux + mv.x;
  164|  79.8k|    const int isuy = by4 * 4 + rsuy;
  165|  79.8k|    const int isux = bx4 * 4 + rsux;
  166|       |
  167|   272k|    for (int i = 0; i < np; i++) {
  ------------------
  |  Branch (167:21): [True: 193k, False: 79.8k]
  ------------------
  168|   193k|        const int dx = pts[i][1][0] - dux;
  169|   193k|        const int dy = pts[i][1][1] - duy;
  170|   193k|        const int sx = pts[i][0][0] - sux;
  171|   193k|        const int sy = pts[i][0][1] - suy;
  172|   193k|        if (abs(sx - dx) < 256 && abs(sy - dy) < 256) {
  ------------------
  |  Branch (172:13): [True: 190k, False: 2.13k]
  |  Branch (172:35): [True: 189k, False: 1.32k]
  ------------------
  173|   189k|            a[0][0] += ((sx * sx) >> 2) + sx * 2 + 8;
  174|   189k|            a[0][1] += ((sx * sy) >> 2) + sx + sy + 4;
  175|   189k|            a[1][1] += ((sy * sy) >> 2) + sy * 2 + 8;
  176|   189k|            bx[0] += ((sx * dx) >> 2) + sx + dx + 8;
  177|   189k|            bx[1] += ((sy * dx) >> 2) + sy + dx + 4;
  178|   189k|            by[0] += ((sx * dy) >> 2) + sx + dy + 4;
  179|   189k|            by[1] += ((sy * dy) >> 2) + sy + dy + 8;
  180|   189k|        }
  181|   193k|    }
  182|       |
  183|       |    // compute determinant of a
  184|  79.8k|    const int64_t det = (int64_t) a[0][0] * a[1][1] - (int64_t) a[0][1] * a[0][1];
  185|  79.8k|    if (det == 0) return 1;
  ------------------
  |  Branch (185:9): [True: 3.46k, False: 76.3k]
  ------------------
  186|  76.3k|    int shift, idet = apply_sign64(resolve_divisor_64(llabs(det), &shift), det);
  187|  76.3k|    shift -= 16;
  188|  76.3k|    if (shift < 0) {
  ------------------
  |  Branch (188:9): [True: 0, False: 76.3k]
  ------------------
  189|      0|        idet <<= -shift;
  190|      0|        shift = 0;
  191|      0|    }
  192|       |
  193|       |    // solve the least-squares
  194|  76.3k|    mat[2] = get_mult_shift_diag((int64_t) a[1][1] * bx[0] -
  195|  76.3k|                                 (int64_t) a[0][1] * bx[1], idet, shift);
  196|  76.3k|    mat[3] = get_mult_shift_ndiag((int64_t) a[0][0] * bx[1] -
  197|  76.3k|                                  (int64_t) a[0][1] * bx[0], idet, shift);
  198|  76.3k|    mat[4] = get_mult_shift_ndiag((int64_t) a[1][1] * by[0] -
  199|  76.3k|                                  (int64_t) a[0][1] * by[1], idet, shift);
  200|  76.3k|    mat[5] = get_mult_shift_diag((int64_t) a[0][0] * by[1] -
  201|  76.3k|                                 (int64_t) a[0][1] * by[0], idet, shift);
  202|       |
  203|  76.3k|    mat[0] = iclip(mv.x * 0x2000 - (isux * (mat[2] - 0x10000) + isuy * mat[3]),
  204|  76.3k|                   -0x800000, 0x7fffff);
  205|  76.3k|    mat[1] = iclip(mv.y * 0x2000 - (isux * mat[4] + isuy * (mat[5] - 0x10000)),
  206|  76.3k|                   -0x800000, 0x7fffff);
  207|       |
  208|  76.3k|    return 0;
  209|  79.8k|}
warpmv.c:iclip_wmp:
   63|   323k|static inline int iclip_wmp(const int v) {
   64|   323k|    const int cv = iclip(v, INT16_MIN, INT16_MAX);
   65|       |
   66|   323k|    return apply_sign((abs(cv) + 32) >> 6, cv) * (1 << 6);
   67|   323k|}
warpmv.c:resolve_divisor_32:
   69|  80.9k|static inline int resolve_divisor_32(const unsigned d, int *const shift) {
   70|  80.9k|    *shift = ulog2(d);
   71|  80.9k|    const int e = d - (1 << *shift);
   72|  80.9k|    const int f = *shift > 8 ? (e + (1 << (*shift - 9))) >> (*shift - 8) :
  ------------------
  |  Branch (72:19): [True: 80.9k, False: 0]
  ------------------
   73|  80.9k|                               e << (8 - *shift);
   74|  80.9k|    assert(f <= 256);
  ------------------
  |  Branch (74:5): [True: 80.9k, False: 0]
  ------------------
   75|  80.9k|    *shift += 14;
   76|       |    // Use f as lookup into the precomputed table of multipliers
   77|  80.9k|    return div_lut[f];
   78|  80.9k|}
warpmv.c:resolve_divisor_64:
  102|  76.3k|static int resolve_divisor_64(const uint64_t d, int *const shift) {
  103|  76.3k|    *shift = u64log2(d);
  104|  76.3k|    const int64_t e = d - (1LL << *shift);
  105|  76.3k|    const int64_t f = *shift > 8 ? (e + (1LL << (*shift - 9))) >> (*shift - 8) :
  ------------------
  |  Branch (105:23): [True: 76.3k, False: 0]
  ------------------
  106|  76.3k|                                   e << (8 - *shift);
  107|  76.3k|    assert(f <= 256);
  ------------------
  |  Branch (107:5): [True: 76.3k, False: 0]
  ------------------
  108|  76.3k|    *shift += 14;
  109|       |    // Use f as lookup into the precomputed table of multipliers
  110|  76.3k|    return div_lut[f];
  111|  76.3k|}
warpmv.c:get_mult_shift_diag:
  125|   152k|{
  126|   152k|    const int64_t v1 = px * idet;
  127|   152k|    const int v2 = apply_sign64((int) ((llabs(v1) +
  128|   152k|                                        ((1LL << shift) >> 1)) >> shift),
  129|   152k|                                v1);
  130|   152k|    return iclip(v2, 0xe001, 0x11fff);
  131|   152k|}
warpmv.c:get_mult_shift_ndiag:
  115|   152k|{
  116|   152k|    const int64_t v1 = px * idet;
  117|   152k|    const int v2 = apply_sign64((int) ((llabs(v1) +
  118|   152k|                                        ((1LL << shift) >> 1)) >> shift),
  119|   152k|                                v1);
  120|   152k|    return iclip(v2, -0x1fff, 0x1fff);
  121|   152k|}

dav1d_init_ii_wedge_masks:
  207|      1|COLD void dav1d_init_ii_wedge_masks(void) {
  208|       |    // This function is guaranteed to be called only once
  209|       |
  210|      1|    enum WedgeMasterLineType {
  211|      1|        WEDGE_MASTER_LINE_ODD,
  212|      1|        WEDGE_MASTER_LINE_EVEN,
  213|      1|        WEDGE_MASTER_LINE_VERT,
  214|      1|        N_WEDGE_MASTER_LINES,
  215|      1|    };
  216|      1|    static const uint8_t wedge_master_border[N_WEDGE_MASTER_LINES][8] = {
  217|      1|        [WEDGE_MASTER_LINE_ODD]  = {  1,  2,  6, 18, 37, 53, 60, 63 },
  218|      1|        [WEDGE_MASTER_LINE_EVEN] = {  1,  4, 11, 27, 46, 58, 62, 63 },
  219|      1|        [WEDGE_MASTER_LINE_VERT] = {  0,  2,  7, 21, 43, 57, 62, 64 },
  220|      1|    };
  221|      1|    uint8_t master[6][64 * 64];
  222|       |
  223|       |    // create master templates
  224|     65|    for (int y = 0, off = 0; y < 64; y++, off += 64)
  ------------------
  |  Branch (224:30): [True: 64, False: 1]
  ------------------
  225|     64|        insert_border(&master[WEDGE_VERTICAL][off],
  226|     64|                      wedge_master_border[WEDGE_MASTER_LINE_VERT], 32);
  227|     33|    for (int y = 0, off = 0, ctr = 48; y < 64; y += 2, off += 128, ctr--)
  ------------------
  |  Branch (227:40): [True: 32, False: 1]
  ------------------
  228|     32|    {
  229|     32|        insert_border(&master[WEDGE_OBLIQUE63][off],
  230|     32|                      wedge_master_border[WEDGE_MASTER_LINE_EVEN], ctr);
  231|     32|        insert_border(&master[WEDGE_OBLIQUE63][off + 64],
  232|     32|                      wedge_master_border[WEDGE_MASTER_LINE_ODD], ctr - 1);
  233|     32|    }
  234|       |
  235|      1|    transpose(master[WEDGE_OBLIQUE27], master[WEDGE_OBLIQUE63]);
  236|      1|    transpose(master[WEDGE_HORIZONTAL], master[WEDGE_VERTICAL]);
  237|      1|    hflip(master[WEDGE_OBLIQUE117], master[WEDGE_OBLIQUE63]);
  238|      1|    hflip(master[WEDGE_OBLIQUE153], master[WEDGE_OBLIQUE27]);
  239|       |
  240|      1|#define fill(w, h, sz_422, sz_420, hvsw, signs) \
  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  242|      1|                master, wedge_codebook_16_##hvsw, \
  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  246|       |
  247|      1|    fill(32, 32, 16x32, 16x16, heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  248|      1|    fill(32, 16, 16x16, 16x8,  hltw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  249|      1|    fill(32,  8, 16x8,  16x4,  hltw, 0x6beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  250|      1|    fill(16, 32,  8x32,  8x16, hgtw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  251|      1|    fill(16, 16,  8x16,  8x8,  heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  252|      1|    fill(16,  8,  8x8,   8x4,  hltw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  253|      1|    fill( 8, 32,  4x32,  4x16, hgtw, 0x7aeb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  254|      1|    fill( 8, 16,  4x16,  4x8,  hgtw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  255|      1|    fill( 8,  8,  4x8,   4x4,  heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  256|      1|#undef fill
  257|       |
  258|      1|    memset(dav1d_masks.ii_dc, 32, 32 * 32);
  259|      4|    for (int c = 0; c < 3; c++) {
  ------------------
  |  Branch (259:21): [True: 3, False: 1]
  ------------------
  260|      3|        dav1d_masks.offsets[c][BS_32x32-BS_32x32].ii[II_DC_PRED] =
  261|      3|        dav1d_masks.offsets[c][BS_32x16-BS_32x32].ii[II_DC_PRED] =
  262|      3|        dav1d_masks.offsets[c][BS_16x32-BS_32x32].ii[II_DC_PRED] =
  263|      3|        dav1d_masks.offsets[c][BS_16x16-BS_32x32].ii[II_DC_PRED] =
  264|      3|        dav1d_masks.offsets[c][BS_16x8 -BS_32x32].ii[II_DC_PRED] =
  265|      3|        dav1d_masks.offsets[c][BS_8x16 -BS_32x32].ii[II_DC_PRED] =
  266|      3|        dav1d_masks.offsets[c][BS_8x8  -BS_32x32].ii[II_DC_PRED] =
  267|      3|            MASK_OFFSET(dav1d_masks.ii_dc);
  ------------------
  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  268|      3|    }
  269|       |
  270|      1|#define BUILD_NONDC_II_MASKS(w, h, step) \
  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  272|       |
  273|      1|#define ASSIGN_NONDC_II_OFFSET(bs, w444, h444, w422, h422, w420, h420) \
  274|      1|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  275|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  276|      1|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  277|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  278|      1|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  279|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  280|       |
  281|      1|    BUILD_NONDC_II_MASKS(32, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  282|      1|    BUILD_NONDC_II_MASKS(16, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  283|      1|    BUILD_NONDC_II_MASKS(16, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  284|      1|    BUILD_NONDC_II_MASKS( 8, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  285|      1|    BUILD_NONDC_II_MASKS( 8, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  286|      1|    BUILD_NONDC_II_MASKS( 8,  8, 4);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  287|      1|    BUILD_NONDC_II_MASKS( 4, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  288|      1|    BUILD_NONDC_II_MASKS( 4,  8, 4);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  289|      1|    BUILD_NONDC_II_MASKS( 4,  4, 8);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  290|      4|    for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (290:21): [True: 3, False: 1]
  ------------------
  291|      3|        ASSIGN_NONDC_II_OFFSET(BS_32x32, 32, 32, 16, 32, 16, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  292|      3|        ASSIGN_NONDC_II_OFFSET(BS_32x16, 32, 32, 16, 16, 16, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  293|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x32, 16, 32,  8, 32,  8, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  294|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x16, 16, 16,  8, 16,  8,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  295|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x8,  16, 16,  8,  8,  8,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  296|      3|        ASSIGN_NONDC_II_OFFSET(BS_8x16,   8, 16,  4, 16,  4,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  297|      3|        ASSIGN_NONDC_II_OFFSET(BS_8x8,    8,  8,  4,  8,  4,  4);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  298|      3|    }
  299|      1|}
wedge.c:insert_border:
   90|    128|{
   91|    128|    if (ctr > 4) memset(dst, 0, ctr - 4);
  ------------------
  |  Branch (91:9): [True: 128, False: 0]
  ------------------
   92|    128|    memcpy(dst + imax(ctr, 4) - 4, src + imax(4 - ctr, 0), imin(64 - ctr, 8));
   93|    128|    if (ctr < 64 - 4)
  ------------------
  |  Branch (93:9): [True: 128, False: 0]
  ------------------
   94|    128|        memset(dst + ctr + 4, 64, 64 - 4 - ctr);
   95|    128|}
wedge.c:transpose:
   97|      2|static void transpose(uint8_t *const dst, const uint8_t *const src) {
   98|    130|    for (int y = 0, y_off = 0; y < 64; y++, y_off += 64)
  ------------------
  |  Branch (98:32): [True: 128, False: 2]
  ------------------
   99|  8.32k|        for (int x = 0, x_off = 0; x < 64; x++, x_off += 64)
  ------------------
  |  Branch (99:36): [True: 8.19k, False: 128]
  ------------------
  100|  8.19k|            dst[x_off + y] = src[y_off + x];
  101|      2|}
wedge.c:hflip:
  103|      2|static void hflip(uint8_t *const dst, const uint8_t *const src) {
  104|    130|    for (int y = 0, y_off = 0; y < 64; y++, y_off += 64)
  ------------------
  |  Branch (104:32): [True: 128, False: 2]
  ------------------
  105|  8.32k|        for (int x = 0; x < 64; x++)
  ------------------
  |  Branch (105:25): [True: 8.19k, False: 128]
  ------------------
  106|  8.19k|            dst[y_off + 64 - 1 - x] = src[y_off + x];
  107|      2|}
wedge.c:fill2d_16x2:
  153|      9|{
  154|      9|    const int n_stride_444 = (w * h);
  155|      9|    const int n_stride_422 = n_stride_444 >> 1;
  156|      9|    const int n_stride_420 = n_stride_444 >> 2;
  157|      9|    const int sign_stride_422 = 16 * n_stride_422;
  158|      9|    const int sign_stride_420 = 16 * n_stride_420;
  159|       |
  160|       |    // assign pointer offsets in lookup table
  161|    153|    for (int n = 0; n < 16; n++) {
  ------------------
  |  Branch (161:21): [True: 144, False: 9]
  ------------------
  162|    144|        const int sign = signs & 1;
  163|       |
  164|    144|        copy2d(masks_444, master[cb[n].direction], sign, w, h,
  165|    144|               32 - (w * cb[n].x_offset >> 3), 32 - (h * cb[n].y_offset >> 3));
  166|       |
  167|       |        // not using !sign is intentional here, since 444 does not require
  168|       |        // any rounding since no chroma subsampling is applied.
  169|    144|        dav1d_masks.offsets[0][bs].wedge[0][n] =
  170|    144|        dav1d_masks.offsets[0][bs].wedge[1][n] = MASK_OFFSET(masks_444);
  ------------------
  |  |  129|    144|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  171|       |
  172|    144|        dav1d_masks.offsets[1][bs].wedge[0][n] =
  173|    144|            init_chroma(&masks_422[ sign * sign_stride_422], masks_444, 0, w, h, 0);
  174|    144|        dav1d_masks.offsets[1][bs].wedge[1][n] =
  175|    144|            init_chroma(&masks_422[!sign * sign_stride_422], masks_444, 1, w, h, 0);
  176|    144|        dav1d_masks.offsets[2][bs].wedge[0][n] =
  177|    144|            init_chroma(&masks_420[ sign * sign_stride_420], masks_444, 0, w, h, 1);
  178|    144|        dav1d_masks.offsets[2][bs].wedge[1][n] =
  179|    144|            init_chroma(&masks_420[!sign * sign_stride_420], masks_444, 1, w, h, 1);
  180|       |
  181|    144|        signs >>= 1;
  182|    144|        masks_444 += n_stride_444;
  183|    144|        masks_422 += n_stride_422;
  184|    144|        masks_420 += n_stride_420;
  185|    144|    }
  186|      9|}
wedge.c:copy2d:
  111|    144|{
  112|    144|    src += y_off * 64 + x_off;
  113|    144|    if (sign) {
  ------------------
  |  Branch (113:9): [True: 109, False: 35]
  ------------------
  114|  2.14k|        for (int y = 0; y < h; y++) {
  ------------------
  |  Branch (114:25): [True: 2.03k, False: 109]
  ------------------
  115|  40.4k|            for (int x = 0; x < w; x++)
  ------------------
  |  Branch (115:29): [True: 38.4k, False: 2.03k]
  ------------------
  116|  38.4k|                dst[x] = 64 - src[x];
  117|  2.03k|            src += 64;
  118|  2.03k|            dst += w;
  119|  2.03k|        }
  120|    109|    } else {
  121|    691|        for (int y = 0; y < h; y++) {
  ------------------
  |  Branch (121:25): [True: 656, False: 35]
  ------------------
  122|    656|            memcpy(dst, src, w);
  123|    656|            src += 64;
  124|    656|            dst += w;
  125|    656|        }
  126|     35|    }
  127|    144|}
wedge.c:init_chroma:
  134|    576|{
  135|    576|    const uint16_t offset = MASK_OFFSET(chroma);
  ------------------
  |  |  129|    576|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  136|  8.64k|    for (int y = 0; y < h; y += 1 + ss_ver) {
  ------------------
  |  Branch (136:21): [True: 8.06k, False: 576]
  ------------------
  137|  83.3k|        for (int x = 0; x < w; x += 2) {
  ------------------
  |  Branch (137:25): [True: 75.2k, False: 8.06k]
  ------------------
  138|  75.2k|            int sum = luma[x] + luma[x + 1] + 1;
  139|  75.2k|            if (ss_ver) sum += luma[w + x] + luma[w + x + 1] + 1;
  ------------------
  |  Branch (139:17): [True: 25.0k, False: 50.1k]
  ------------------
  140|  75.2k|            chroma[x >> 1] = (sum - sign) >> (1 + ss_ver);
  141|  75.2k|        }
  142|  8.06k|        luma += w << ss_ver;
  143|  8.06k|        chroma += w >> 1;
  144|  8.06k|    }
  145|    576|    return offset;
  146|    576|}
wedge.c:build_nondc_ii_masks:
  190|      9|{
  191|      9|    static const uint8_t ii_weights_1d[32] = {
  192|      9|        60, 52, 45, 39, 34, 30, 26, 22, 19, 17, 15, 13, 11, 10,  8,  7,
  193|      9|         6,  6,  5,  4,  4,  3,  3,  2,  2,  2,  2,  1,  1,  1,  1,  1,
  194|      9|    };
  195|       |
  196|      9|    uint8_t *const mask_h  = &mask_v[w * h];
  197|      9|    uint8_t *const mask_sm = &mask_h[w * h];
  198|    173|    for (int y = 0, off = 0; y < h; y++, off += w) {
  ------------------
  |  Branch (198:30): [True: 164, False: 9]
  ------------------
  199|    164|        memset(&mask_v[off], ii_weights_1d[y * step], w);
  200|  2.51k|        for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (200:25): [True: 2.35k, False: 164]
  ------------------
  201|  2.35k|            mask_sm[off + x] = ii_weights_1d[imin(x, y) * step];
  202|  2.35k|            mask_h[off + x] = ii_weights_1d[x * step];
  203|  2.35k|        }
  204|    164|    }
  205|      9|}

cdef_tmpl.c:cdef_dsp_init_x86:
   46|  3.66k|static ALWAYS_INLINE void cdef_dsp_init_x86(Dav1dCdefDSPContext *const c) {
   47|  3.66k|    const unsigned flags = dav1d_get_cpu_flags();
   48|       |
   49|  3.66k|#if BITDEPTH == 8
   50|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (50:9): [True: 0, False: 3.66k]
  ------------------
   51|       |
   52|  3.66k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, sse2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   53|  3.66k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, sse2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   54|  3.66k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, sse2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  3.66k|#endif
   56|       |
   57|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (57:9): [True: 0, False: 3.66k]
  ------------------
   58|       |
   59|  3.66k|    c->dir = BF(dav1d_cdef_dir, ssse3);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   60|  3.66k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, ssse3);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   61|  3.66k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, ssse3);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|  3.66k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, ssse3);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|       |
   64|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (64:9): [True: 0, False: 3.66k]
  ------------------
   65|       |
   66|  3.66k|    c->dir = BF(dav1d_cdef_dir, sse4);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   67|  3.66k|#if BITDEPTH == 8
   68|  3.66k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, sse4);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   69|  3.66k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, sse4);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   70|  3.66k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, sse4);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   71|  3.66k|#endif
   72|       |
   73|  3.66k|#if ARCH_X86_64
   74|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (74:9): [True: 0, False: 3.66k]
  ------------------
   75|       |
   76|  3.66k|    c->dir = BF(dav1d_cdef_dir, avx2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   77|  3.66k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, avx2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   78|  3.66k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, avx2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   79|  3.66k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, avx2);
  ------------------
  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   80|       |
   81|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (81:9): [True: 3.66k, False: 0]
  ------------------
   82|       |
   83|      0|    c->fb[0] = BF(dav1d_cdef_filter_8x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   84|      0|    c->fb[1] = BF(dav1d_cdef_filter_4x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   85|      0|    c->fb[2] = BF(dav1d_cdef_filter_4x4, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   86|      0|#endif
   87|      0|}

dav1d_get_cpu_flags_x86:
   47|      1|COLD unsigned dav1d_get_cpu_flags_x86(void) {
   48|      1|    union {
   49|      1|        CpuidRegisters r;
   50|      1|        struct {
   51|      1|            uint32_t max_leaf;
   52|      1|            char vendor[12];
   53|      1|        };
   54|      1|    } cpu;
   55|      1|    dav1d_cpu_cpuid(&cpu.r, 0, 0);
   56|      1|    unsigned flags = dav1d_get_default_cpu_flags();
   57|       |
   58|      1|    if (cpu.max_leaf >= 1) {
  ------------------
  |  Branch (58:9): [True: 1, False: 0]
  ------------------
   59|      1|        CpuidRegisters r;
   60|      1|        dav1d_cpu_cpuid(&r, 1, 0);
   61|      1|        const unsigned family = ((r.eax >> 8) & 0x0f) + ((r.eax >> 20) & 0xff);
   62|       |
   63|      1|        if (X(r.edx, 0x06008000)) /* CMOV/SSE/SSE2 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   64|      1|            flags |= DAV1D_X86_CPU_FLAG_SSE2;
   65|      1|            if (X(r.ecx, 0x00000201)) /* SSE3/SSSE3 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   66|      1|                flags |= DAV1D_X86_CPU_FLAG_SSSE3;
   67|      1|                if (X(r.ecx, 0x00080000)) /* SSE4.1 */
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   68|      1|                    flags |= DAV1D_X86_CPU_FLAG_SSE41;
   69|      1|            }
   70|      1|        }
   71|      1|#if ARCH_X86_64
   72|       |        /* We only support >128-bit SIMD on x86-64. */
   73|      1|        if (X(r.ecx, 0x18000000)) /* OSXSAVE/AVX */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   74|      1|            const uint64_t xcr0 = dav1d_cpu_xgetbv(0);
   75|      1|            if (X(xcr0, 0x00000006)) /* XMM/YMM */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   76|      1|                if (cpu.max_leaf >= 7) {
  ------------------
  |  Branch (76:21): [True: 1, False: 0]
  ------------------
   77|      1|                    dav1d_cpu_cpuid(&r, 7, 0);
   78|      1|                    if (X(r.ebx, 0x00000128)) /* BMI1/BMI2/AVX2 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   79|      1|                        flags |= DAV1D_X86_CPU_FLAG_AVX2;
   80|      1|                        if (X(xcr0, 0x000000e0)) /* ZMM/OPMASK */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 1]
  |  |  ------------------
  ------------------
   81|      0|                            if (X(r.ebx, 0xd0230000) && X(r.ecx, 0x00005f42))
  ------------------
  |  |   45|      0|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (X(r.ebx, 0xd0230000) && X(r.ecx, 0x00005f42))
  ------------------
  |  |   45|      0|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 0]
  |  |  ------------------
  ------------------
   82|      0|                                flags |= DAV1D_X86_CPU_FLAG_AVX512ICL;
   83|      0|                        }
   84|      1|                    }
   85|      1|                }
   86|      1|            }
   87|      1|        }
   88|      1|#endif
   89|      1|        if (!memcmp(cpu.vendor, "AuthenticAMD", sizeof(cpu.vendor))) {
  ------------------
  |  Branch (89:13): [True: 1, False: 0]
  ------------------
   90|      1|            if ((flags & DAV1D_X86_CPU_FLAG_AVX2) && family <= 0x19) {
  ------------------
  |  Branch (90:17): [True: 1, False: 0]
  |  Branch (90:54): [True: 1, False: 0]
  ------------------
   91|       |                /* Excavator, Zen, Zen+, Zen 2, Zen 3, Zen 3+, Zen 4 */
   92|      1|                flags |= DAV1D_X86_CPU_FLAG_SLOW_GATHER;
   93|      1|            }
   94|      1|        }
   95|      1|    }
   96|       |
   97|      1|    return flags;
   98|      1|}

filmgrain_tmpl.c:film_grain_dsp_init_x86:
   45|  8.66k|static ALWAYS_INLINE void film_grain_dsp_init_x86(Dav1dFilmGrainDSPContext *const c) {
   46|  8.66k|    const unsigned flags = dav1d_get_cpu_flags();
   47|       |
   48|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (48:9): [True: 0, False: 8.66k]
  ------------------
   49|       |
   50|  8.66k|    c->generate_grain_y = BF(dav1d_generate_grain_y, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   51|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_generate_grain_uv_420, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   52|  8.66k|    c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   53|  8.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   54|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_generate_grain_uv_422, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_generate_grain_uv_444, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.66k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   58|       |
   59|  8.66k|#if ARCH_X86_64
   60|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (60:9): [True: 0, False: 8.66k]
  ------------------
   61|       |
   62|  8.66k|    c->generate_grain_y = BF(dav1d_generate_grain_y, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_generate_grain_uv_420, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   64|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_generate_grain_uv_422, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   65|  8.66k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_generate_grain_uv_444, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|       |
   67|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (67:9): [True: 0, False: 8.66k]
  ------------------
   68|      0|        c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   69|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   70|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   71|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   72|      0|    }
   73|       |
   74|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (74:9): [True: 8.66k, False: 0]
  ------------------
   75|       |
   76|      0|    if (BITDEPTH == 8 || !(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (76:9): [True: 0, Folded]
  |  Branch (76:26): [True: 0, False: 0]
  ------------------
   77|      0|        c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   78|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   79|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   80|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   81|      0|    }
   82|      0|#endif
   83|      0|}

ipred_tmpl.c:intra_pred_dsp_init_x86:
   71|  8.66k|static ALWAYS_INLINE void intra_pred_dsp_init_x86(Dav1dIntraPredDSPContext *const c) {
   72|  8.66k|    const unsigned flags = dav1d_get_cpu_flags();
   73|       |
   74|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (74:9): [True: 0, False: 8.66k]
  ------------------
   75|       |
   76|  8.66k|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   77|  8.66k|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   78|  8.66k|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   79|  8.66k|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   80|  8.66k|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   81|  8.66k|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   82|  8.66k|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   83|  8.66k|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   84|  8.66k|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   85|  8.66k|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   86|  8.66k|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   87|  8.66k|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   88|  8.66k|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   89|  8.66k|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   ssse3);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   90|       |
   91|  8.66k|    init_cfl_pred_fn(DC_PRED,      ipred_cfl,      ssse3);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   92|  8.66k|    init_cfl_pred_fn(DC_128_PRED,  ipred_cfl_128,  ssse3);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   93|  8.66k|    init_cfl_pred_fn(TOP_DC_PRED,  ipred_cfl_top,  ssse3);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   94|  8.66k|    init_cfl_pred_fn(LEFT_DC_PRED, ipred_cfl_left, ssse3);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   95|       |
   96|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I420 - 1, ipred_cfl_ac_420, ssse3);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   97|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I422 - 1, ipred_cfl_ac_422, ssse3);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   98|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I444 - 1, ipred_cfl_ac_444, ssse3);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   99|       |
  100|  8.66k|    c->pal_pred = BF(dav1d_pal_pred, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  101|       |
  102|  8.66k|#if ARCH_X86_64
  103|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (103:9): [True: 0, False: 8.66k]
  ------------------
  104|       |
  105|  8.66k|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  106|  8.66k|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  107|  8.66k|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  108|  8.66k|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  109|  8.66k|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  110|  8.66k|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  111|  8.66k|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  112|  8.66k|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  113|  8.66k|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  114|  8.66k|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  115|  8.66k|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  116|  8.66k|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  117|  8.66k|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  118|  8.66k|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   avx2);
  ------------------
  |  |   39|  8.66k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  119|       |
  120|  8.66k|    init_cfl_pred_fn(DC_PRED,      ipred_cfl,      avx2);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  121|  8.66k|    init_cfl_pred_fn(DC_128_PRED,  ipred_cfl_128,  avx2);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  122|  8.66k|    init_cfl_pred_fn(TOP_DC_PRED,  ipred_cfl_top,  avx2);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  123|  8.66k|    init_cfl_pred_fn(LEFT_DC_PRED, ipred_cfl_left, avx2);
  ------------------
  |  |   41|  8.66k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  124|       |
  125|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I420 - 1, ipred_cfl_ac_420, avx2);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  126|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I422 - 1, ipred_cfl_ac_422, avx2);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  127|  8.66k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I444 - 1, ipred_cfl_ac_444, avx2);
  ------------------
  |  |   43|  8.66k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  128|       |
  129|  8.66k|    c->pal_pred = BF(dav1d_pal_pred, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  130|       |
  131|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (131:9): [True: 8.66k, False: 0]
  ------------------
  132|       |
  133|      0|#if BITDEPTH == 8
  134|      0|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.66k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  135|      0|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  136|      0|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  137|      0|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  138|      0|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  139|      0|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  140|      0|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  141|      0|#endif
  142|      0|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  143|      0|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  144|      0|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  145|      0|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  146|      0|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  147|      0|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  148|      0|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  149|      0|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  150|       |
  151|      0|    c->pal_pred = BF(dav1d_pal_pred, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  152|      0|#endif
  153|      0|}

itx_tmpl.c:itx_dsp_init_x86:
  112|  3.66k|{
  113|  3.66k|#define assign_itx_bpc_fn(pfx, w, h, type, type_enum, bpc, ext) \
  114|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  115|  3.66k|        BF_BPC(dav1d_inv_txfm_add_##type##_##w##x##h, bpc, ext)
  116|       |
  117|  3.66k|#define assign_itx1_bpc_fn(pfx, w, h, bpc, ext) \
  118|  3.66k|    assign_itx_bpc_fn(pfx, w, h, dct_dct,           DCT_DCT,           bpc, ext)
  119|       |
  120|  3.66k|#define assign_itx2_bpc_fn(pfx, w, h, bpc, ext) \
  121|  3.66k|    assign_itx1_bpc_fn(pfx, w, h, bpc, ext); \
  122|  3.66k|    assign_itx_bpc_fn(pfx, w, h, identity_identity, IDTX,              bpc, ext)
  123|       |
  124|  3.66k|#define assign_itx12_bpc_fn(pfx, w, h, bpc, ext) \
  125|  3.66k|    assign_itx2_bpc_fn(pfx, w, h, bpc, ext); \
  126|  3.66k|    assign_itx_bpc_fn(pfx, w, h, dct_adst,          ADST_DCT,          bpc, ext); \
  127|  3.66k|    assign_itx_bpc_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      bpc, ext); \
  128|  3.66k|    assign_itx_bpc_fn(pfx, w, h, dct_identity,      H_DCT,             bpc, ext); \
  129|  3.66k|    assign_itx_bpc_fn(pfx, w, h, adst_dct,          DCT_ADST,          bpc, ext); \
  130|  3.66k|    assign_itx_bpc_fn(pfx, w, h, adst_adst,         ADST_ADST,         bpc, ext); \
  131|  3.66k|    assign_itx_bpc_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     bpc, ext); \
  132|  3.66k|    assign_itx_bpc_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      bpc, ext); \
  133|  3.66k|    assign_itx_bpc_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     bpc, ext); \
  134|  3.66k|    assign_itx_bpc_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, bpc, ext); \
  135|  3.66k|    assign_itx_bpc_fn(pfx, w, h, identity_dct,      V_DCT,             bpc, ext)
  136|       |
  137|  3.66k|#define assign_itx16_bpc_fn(pfx, w, h, bpc, ext) \
  138|  3.66k|    assign_itx12_bpc_fn(pfx, w, h, bpc, ext); \
  139|  3.66k|    assign_itx_bpc_fn(pfx, w, h, adst_identity,     H_ADST,            bpc, ext); \
  140|  3.66k|    assign_itx_bpc_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        bpc, ext); \
  141|  3.66k|    assign_itx_bpc_fn(pfx, w, h, identity_adst,     V_ADST,            bpc, ext); \
  142|  3.66k|    assign_itx_bpc_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        bpc, ext)
  143|       |
  144|  3.66k|    const unsigned flags = dav1d_get_cpu_flags();
  145|       |
  146|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (146:9): [True: 0, False: 3.66k]
  ------------------
  147|       |
  148|  3.66k|    assign_itx_fn(, 4, 4, wht_wht, WHT_WHT, sse2);
  ------------------
  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  ------------------
  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  149|       |
  150|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (150:9): [True: 0, False: 3.66k]
  ------------------
  151|       |
  152|  3.66k|#if BITDEPTH == 8
  153|  3.66k|    assign_itx16_fn(,   4,  4, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  154|  3.66k|    assign_itx16_fn(R,  4,  8, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  155|  3.66k|    assign_itx16_fn(R,  8,  4, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  156|  3.66k|    assign_itx16_fn(,   8,  8, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  157|  3.66k|    assign_itx16_fn(R,  4, 16, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  158|  3.66k|    assign_itx16_fn(R, 16,  4, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  159|  3.66k|    assign_itx16_fn(R,  8, 16, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  160|  3.66k|    assign_itx16_fn(R, 16,  8, ssse3);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  161|  3.66k|    assign_itx12_fn(,  16, 16, ssse3);
  ------------------
  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  162|  3.66k|    assign_itx2_fn (R,  8, 32, ssse3);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  163|  3.66k|    assign_itx2_fn (R, 32,  8, ssse3);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  164|  3.66k|    assign_itx2_fn (R, 16, 32, ssse3);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  165|  3.66k|    assign_itx2_fn (R, 32, 16, ssse3);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  166|  3.66k|    assign_itx2_fn (,  32, 32, ssse3);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  167|  3.66k|    assign_itx1_fn (R, 16, 64, ssse3);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  168|  3.66k|    assign_itx1_fn (R, 32, 64, ssse3);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  169|  3.66k|    assign_itx1_fn (R, 64, 16, ssse3);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  170|  3.66k|    assign_itx1_fn (R, 64, 32, ssse3);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  171|  3.66k|    assign_itx1_fn ( , 64, 64, ssse3);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  172|  3.66k|    *all_simd = 1;
  173|  3.66k|#endif
  174|       |
  175|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (175:9): [True: 0, False: 3.66k]
  ------------------
  176|       |
  177|       |#if BITDEPTH == 16
  178|       |    if (bpc == 10) {
  179|       |        assign_itx16_fn(,   4,  4, sse4);
  180|       |        assign_itx16_fn(R,  4,  8, sse4);
  181|       |        assign_itx16_fn(R,  4, 16, sse4);
  182|       |        assign_itx16_fn(R,  8,  4, sse4);
  183|       |        assign_itx16_fn(,   8,  8, sse4);
  184|       |        assign_itx16_fn(R,  8, 16, sse4);
  185|       |        assign_itx16_fn(R, 16,  4, sse4);
  186|       |        assign_itx16_fn(R, 16,  8, sse4);
  187|       |        assign_itx12_fn(,  16, 16, sse4);
  188|       |        assign_itx2_fn (R,  8, 32, sse4);
  189|       |        assign_itx2_fn (R, 32,  8, sse4);
  190|       |        assign_itx2_fn (R, 16, 32, sse4);
  191|       |        assign_itx2_fn (R, 32, 16, sse4);
  192|       |        assign_itx2_fn (,  32, 32, sse4);
  193|       |        assign_itx1_fn (R, 16, 64, sse4);
  194|       |        assign_itx1_fn (R, 32, 64, sse4);
  195|       |        assign_itx1_fn (R, 64, 16, sse4);
  196|       |        assign_itx1_fn (R, 64, 32, sse4);
  197|       |        assign_itx1_fn (,  64, 64, sse4);
  198|       |        *all_simd = 1;
  199|       |    }
  200|       |#endif
  201|       |
  202|  3.66k|#if ARCH_X86_64
  203|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (203:9): [True: 0, False: 3.66k]
  ------------------
  204|       |
  205|  3.66k|    assign_itx_fn(, 4, 4, wht_wht, WHT_WHT, avx2);
  ------------------
  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  ------------------
  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  206|       |
  207|  3.66k|#if BITDEPTH == 8
  208|  3.66k|    assign_itx16_fn( ,  4,  4, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  209|  3.66k|    assign_itx16_fn(R,  4,  8, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  210|  3.66k|    assign_itx16_fn(R,  4, 16, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  211|  3.66k|    assign_itx16_fn(R,  8,  4, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  212|  3.66k|    assign_itx16_fn( ,  8,  8, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  213|  3.66k|    assign_itx16_fn(R,  8, 16, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  214|  3.66k|    assign_itx2_fn (R,  8, 32, avx2);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  215|  3.66k|    assign_itx16_fn(R, 16,  4, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  216|  3.66k|    assign_itx16_fn(R, 16,  8, avx2);
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.66k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.66k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  217|  3.66k|    assign_itx12_fn( , 16, 16, avx2);
  ------------------
  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|  3.66k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|  3.66k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|  3.66k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|  3.66k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|  3.66k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|  3.66k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|  3.66k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  218|  3.66k|    assign_itx2_fn (R, 16, 32, avx2);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  219|  3.66k|    assign_itx1_fn (R, 16, 64, avx2);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  220|  3.66k|    assign_itx2_fn (R, 32,  8, avx2);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  221|  3.66k|    assign_itx2_fn (R, 32, 16, avx2);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  222|  3.66k|    assign_itx2_fn ( , 32, 32, avx2);
  ------------------
  |  |   84|  3.66k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  223|  3.66k|    assign_itx1_fn (R, 32, 64, avx2);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  224|  3.66k|    assign_itx1_fn (R, 64, 16, avx2);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  225|  3.66k|    assign_itx1_fn (R, 64, 32, avx2);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  226|  3.66k|    assign_itx1_fn ( , 64, 64, avx2);
  ------------------
  |  |   81|  3.66k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  227|       |#else
  228|       |    if (bpc == 10) {
  229|       |        assign_itx16_bpc_fn( ,  4,  4, 10, avx2);
  230|       |        assign_itx16_bpc_fn(R,  4,  8, 10, avx2);
  231|       |        assign_itx16_bpc_fn(R,  4, 16, 10, avx2);
  232|       |        assign_itx16_bpc_fn(R,  8,  4, 10, avx2);
  233|       |        assign_itx16_bpc_fn( ,  8,  8, 10, avx2);
  234|       |        assign_itx16_bpc_fn(R,  8, 16, 10, avx2);
  235|       |        assign_itx2_bpc_fn (R,  8, 32, 10, avx2);
  236|       |        assign_itx16_bpc_fn(R, 16,  4, 10, avx2);
  237|       |        assign_itx16_bpc_fn(R, 16,  8, 10, avx2);
  238|       |        assign_itx12_bpc_fn( , 16, 16, 10, avx2);
  239|       |        assign_itx2_bpc_fn (R, 16, 32, 10, avx2);
  240|       |        assign_itx1_bpc_fn (R, 16, 64, 10, avx2);
  241|       |        assign_itx2_bpc_fn (R, 32,  8, 10, avx2);
  242|       |        assign_itx2_bpc_fn (R, 32, 16, 10, avx2);
  243|       |        assign_itx2_bpc_fn ( , 32, 32, 10, avx2);
  244|       |        assign_itx1_bpc_fn (R, 32, 64, 10, avx2);
  245|       |        assign_itx1_bpc_fn (R, 64, 16, 10, avx2);
  246|       |        assign_itx1_bpc_fn (R, 64, 32, 10, avx2);
  247|       |        assign_itx1_bpc_fn ( , 64, 64, 10, avx2);
  248|       |    } else {
  249|       |        assign_itx16_bpc_fn( ,  4,  4, 12, avx2);
  250|       |        assign_itx16_bpc_fn(R,  4,  8, 12, avx2);
  251|       |        assign_itx16_bpc_fn(R,  4, 16, 12, avx2);
  252|       |        assign_itx16_bpc_fn(R,  8,  4, 12, avx2);
  253|       |        assign_itx16_bpc_fn( ,  8,  8, 12, avx2);
  254|       |        assign_itx16_bpc_fn(R,  8, 16, 12, avx2);
  255|       |        assign_itx2_bpc_fn (R,  8, 32, 12, avx2);
  256|       |        assign_itx16_bpc_fn(R, 16,  4, 12, avx2);
  257|       |        assign_itx16_bpc_fn(R, 16,  8, 12, avx2);
  258|       |        assign_itx12_bpc_fn( , 16, 16, 12, avx2);
  259|       |        assign_itx2_bpc_fn (R, 32,  8, 12, avx2);
  260|       |        assign_itx_bpc_fn(R, 16, 32, identity_identity, IDTX, 12, avx2);
  261|       |        assign_itx_bpc_fn(R, 32, 16, identity_identity, IDTX, 12, avx2);
  262|       |        assign_itx_bpc_fn( , 32, 32, identity_identity, IDTX, 12, avx2);
  263|       |    }
  264|       |#endif
  265|       |
  266|  3.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (266:9): [True: 3.66k, False: 0]
  ------------------
  267|       |
  268|      0|#if BITDEPTH == 8
  269|  3.66k|    assign_itx16_fn( ,  4,  4, avx512icl); // no wht
  ------------------
  |  |  101|  3.66k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.66k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.66k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.66k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.66k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.66k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.66k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  270|      0|    assign_itx16_fn(R,  4,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  271|      0|    assign_itx16_fn(R,  4, 16, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  272|      0|    assign_itx16_fn(R,  8,  4, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  273|      0|    assign_itx16_fn( ,  8,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  274|      0|    assign_itx16_fn(R,  8, 16, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  275|      0|    assign_itx2_fn (R,  8, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  276|      0|    assign_itx16_fn(R, 16,  4, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  277|      0|    assign_itx16_fn(R, 16,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  278|      0|    assign_itx12_fn( , 16, 16, avx512icl);
  ------------------
  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  279|      0|    assign_itx2_fn (R, 16, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  280|      0|    assign_itx1_fn (R, 16, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  281|      0|    assign_itx2_fn (R, 32,  8, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  282|      0|    assign_itx2_fn (R, 32, 16, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  283|      0|    assign_itx2_fn ( , 32, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  284|      0|    assign_itx1_fn (R, 32, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  285|      0|    assign_itx1_fn (R, 64, 16, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  286|      0|    assign_itx1_fn (R, 64, 32, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  287|      0|    assign_itx1_fn ( , 64, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  288|       |#else
  289|       |    if (bpc == 10) {
  290|       |        assign_itx16_bpc_fn( ,  8,  8, 10, avx512icl);
  291|       |        assign_itx16_bpc_fn(R,  8, 16, 10, avx512icl);
  292|       |        assign_itx2_bpc_fn (R,  8, 32, 10, avx512icl);
  293|       |        assign_itx16_bpc_fn(R, 16,  8, 10, avx512icl);
  294|       |        assign_itx12_bpc_fn( , 16, 16, 10, avx512icl);
  295|       |        assign_itx2_bpc_fn (R, 16, 32, 10, avx512icl);
  296|       |        assign_itx2_bpc_fn (R, 32,  8, 10, avx512icl);
  297|       |        assign_itx2_bpc_fn (R, 32, 16, 10, avx512icl);
  298|       |        assign_itx2_bpc_fn ( , 32, 32, 10, avx512icl);
  299|       |        assign_itx1_bpc_fn (R, 16, 64, 10, avx512icl);
  300|       |        assign_itx1_bpc_fn (R, 32, 64, 10, avx512icl);
  301|       |        assign_itx1_bpc_fn (R, 64, 16, 10, avx512icl);
  302|       |        assign_itx1_bpc_fn (R, 64, 32, 10, avx512icl);
  303|       |        assign_itx1_bpc_fn ( , 64, 64, 10, avx512icl);
  304|       |    }
  305|       |#endif
  306|      0|#endif
  307|      0|}

loopfilter_tmpl.c:loop_filter_dsp_init_x86:
   41|  8.66k|static ALWAYS_INLINE void loop_filter_dsp_init_x86(Dav1dLoopFilterDSPContext *const c) {
   42|  8.66k|    const unsigned flags = dav1d_get_cpu_flags();
   43|       |
   44|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (44:9): [True: 0, False: 8.66k]
  ------------------
   45|       |
   46|  8.66k|    c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   47|  8.66k|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   48|  8.66k|    c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   49|  8.66k|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   50|       |
   51|  8.66k|#if ARCH_X86_64
   52|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (52:9): [True: 0, False: 8.66k]
  ------------------
   53|       |
   54|  8.66k|    c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  8.66k|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.66k|    c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.66k|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   58|       |
   59|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (59:9): [True: 8.66k, False: 0]
  ------------------
   60|       |
   61|      0|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|      0|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|       |
   64|      0|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (64:9): [True: 0, False: 0]
  ------------------
   65|      0|        c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|      0|        c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   67|      0|    }
   68|      0|#endif
   69|      0|}

looprestoration_tmpl.c:loop_restoration_dsp_init_x86:
   50|  8.66k|static ALWAYS_INLINE void loop_restoration_dsp_init_x86(Dav1dLoopRestorationDSPContext *const c, const int bpc) {
   51|  8.66k|    const unsigned flags = dav1d_get_cpu_flags();
   52|       |
   53|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (53:9): [True: 0, False: 8.66k]
  ------------------
   54|  8.66k|#if BITDEPTH == 8
   55|  8.66k|    c->wiener[0] = BF(dav1d_wiener_filter7, sse2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.66k|    c->wiener[1] = BF(dav1d_wiener_filter5, sse2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.66k|#endif
   58|       |
   59|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (59:9): [True: 0, False: 8.66k]
  ------------------
   60|  8.66k|    c->wiener[0] = BF(dav1d_wiener_filter7, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   61|  8.66k|    c->wiener[1] = BF(dav1d_wiener_filter5, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|  8.66k|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (62:9): [True: 3.66k, Folded]
  |  Branch (62:26): [True: 2.47k, False: 2.52k]
  ------------------
   63|  6.14k|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, ssse3);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   64|  6.14k|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, ssse3);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   65|  6.14k|        c->sgr[2] = BF(dav1d_sgr_filter_mix, ssse3);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|  6.14k|    }
   67|       |
   68|  8.66k|#if ARCH_X86_64
   69|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (69:9): [True: 0, False: 8.66k]
  ------------------
   70|       |
   71|  8.66k|    c->wiener[0] = BF(dav1d_wiener_filter7, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   72|  8.66k|    c->wiener[1] = BF(dav1d_wiener_filter5, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   73|  8.66k|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (73:9): [True: 3.66k, Folded]
  |  Branch (73:26): [True: 2.47k, False: 2.52k]
  ------------------
   74|  6.14k|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, avx2);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   75|  6.14k|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, avx2);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   76|  6.14k|        c->sgr[2] = BF(dav1d_sgr_filter_mix, avx2);
  ------------------
  |  |   52|  6.14k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   77|  6.14k|    }
   78|       |
   79|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (79:9): [True: 8.66k, False: 0]
  ------------------
   80|       |
   81|      0|    c->wiener[0] = BF(dav1d_wiener_filter7, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   82|      0|#if BITDEPTH == 8
   83|       |    /* With VNNI we don't need a 5-tap version. */
   84|      0|    c->wiener[1] = c->wiener[0];
   85|       |#else
   86|       |    c->wiener[1] = BF(dav1d_wiener_filter5, avx512icl);
   87|       |#endif
   88|      0|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (88:9): [True: 0, Folded]
  |  Branch (88:26): [True: 0, False: 0]
  ------------------
   89|      0|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   90|      0|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   91|      0|        c->sgr[2] = BF(dav1d_sgr_filter_mix, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   92|      0|    }
   93|      0|#endif
   94|      0|}

mc_tmpl.c:mc_dsp_init_x86:
   92|  8.66k|static ALWAYS_INLINE void mc_dsp_init_x86(Dav1dMCDSPContext *const c) {
   93|  8.66k|    const unsigned flags = dav1d_get_cpu_flags();
   94|       |
   95|  8.66k|    if(!(flags & DAV1D_X86_CPU_FLAG_SSSE3))
  ------------------
  |  Branch (95:8): [True: 0, False: 8.66k]
  ------------------
   96|      0|        return;
   97|       |
   98|  8.66k|    init_8tap_fns(ssse3);
  ------------------
  |  |  143|  8.66k|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|  8.66k|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   99|       |
  100|  8.66k|    init_mc_fn(FILTER_2D_BILINEAR,             bilin,               ssse3);
  ------------------
  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  101|  8.66k|    init_mct_fn(FILTER_2D_BILINEAR,            bilin,               ssse3);
  ------------------
  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  102|       |
  103|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  104|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  105|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  106|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  107|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  108|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  109|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  110|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  111|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  112|  8.66k|    init_mc_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               ssse3);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  113|       |
  114|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  115|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  116|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  117|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  118|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  119|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  120|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  121|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  122|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  123|  8.66k|    init_mct_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               ssse3);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  124|       |
  125|  8.66k|    c->avg = BF(dav1d_avg, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  126|  8.66k|    c->w_avg = BF(dav1d_w_avg, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  127|  8.66k|    c->mask = BF(dav1d_mask, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  128|  8.66k|    c->w_mask[0] = BF(dav1d_w_mask_444, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  129|  8.66k|    c->w_mask[1] = BF(dav1d_w_mask_422, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  130|  8.66k|    c->w_mask[2] = BF(dav1d_w_mask_420, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  131|  8.66k|    c->blend = BF(dav1d_blend, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  132|  8.66k|    c->blend_v = BF(dav1d_blend_v, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  133|  8.66k|    c->blend_h = BF(dav1d_blend_h, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  134|  8.66k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  135|  8.66k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  136|  8.66k|    c->emu_edge = BF(dav1d_emu_edge, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  137|  8.66k|    c->resize = BF(dav1d_resize, ssse3);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  138|       |
  139|  8.66k|    if(!(flags & DAV1D_X86_CPU_FLAG_SSE41))
  ------------------
  |  Branch (139:8): [True: 0, False: 8.66k]
  ------------------
  140|      0|        return;
  141|       |
  142|  8.66k|#if BITDEPTH == 8
  143|  8.66k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, sse4);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  144|  8.66k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, sse4);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  145|  8.66k|#endif
  146|       |
  147|  8.66k|#if ARCH_X86_64
  148|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2))
  ------------------
  |  Branch (148:9): [True: 0, False: 8.66k]
  ------------------
  149|      0|        return;
  150|       |
  151|  8.66k|    init_8tap_fns(avx2);
  ------------------
  |  |  143|  8.66k|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|  8.66k|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  152|       |
  153|  8.66k|    init_mc_fn(FILTER_2D_BILINEAR,            bilin,               avx2);
  ------------------
  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  154|  8.66k|    init_mct_fn(FILTER_2D_BILINEAR,           bilin,               avx2);
  ------------------
  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  155|       |
  156|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  157|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  158|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  159|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  160|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  161|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  162|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  163|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  164|  8.66k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  165|  8.66k|    init_mc_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               avx2);
  ------------------
  |  |   40|  8.66k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  166|       |
  167|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  168|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  169|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  170|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  171|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  172|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  173|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  174|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  175|  8.66k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  176|  8.66k|    init_mct_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               avx2);
  ------------------
  |  |   42|  8.66k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  177|       |
  178|  8.66k|    c->avg = BF(dav1d_avg, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  179|  8.66k|    c->w_avg = BF(dav1d_w_avg, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  180|  8.66k|    c->mask = BF(dav1d_mask, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  181|  8.66k|    c->w_mask[0] = BF(dav1d_w_mask_444, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  182|  8.66k|    c->w_mask[1] = BF(dav1d_w_mask_422, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  183|  8.66k|    c->w_mask[2] = BF(dav1d_w_mask_420, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  184|  8.66k|    c->blend = BF(dav1d_blend, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  185|  8.66k|    c->blend_v = BF(dav1d_blend_v, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  186|  8.66k|    c->blend_h = BF(dav1d_blend_h, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  187|  8.66k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  188|  8.66k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  189|  8.66k|    c->emu_edge = BF(dav1d_emu_edge, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  190|  8.66k|    c->resize = BF(dav1d_resize, avx2);
  ------------------
  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  191|       |
  192|  8.66k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL))
  ------------------
  |  Branch (192:9): [True: 8.66k, False: 0]
  ------------------
  193|  8.66k|        return;
  194|       |
  195|  8.66k|    init_8tap_fns(avx512icl);
  ------------------
  |  |  143|      0|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|      0|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|      0|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.66k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|      0|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.66k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.66k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.66k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  196|       |
  197|      0|    init_mc_fn (FILTER_2D_BILINEAR,            bilin,               avx512icl);
  ------------------
  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  198|      0|    init_mct_fn(FILTER_2D_BILINEAR,            bilin,               avx512icl);
  ------------------
  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  199|       |
  200|      0|    c->avg = BF(dav1d_avg, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  201|      0|    c->w_avg = BF(dav1d_w_avg, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  202|      0|    c->mask = BF(dav1d_mask, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  203|      0|    c->w_mask[0] = BF(dav1d_w_mask_444, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  204|      0|    c->w_mask[1] = BF(dav1d_w_mask_422, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  205|      0|    c->w_mask[2] = BF(dav1d_w_mask_420, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  206|      0|    c->blend = BF(dav1d_blend, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  207|      0|    c->blend_v = BF(dav1d_blend_v, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  208|      0|    c->blend_h = BF(dav1d_blend_h, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  209|       |
  210|      0|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (210:9): [True: 0, False: 0]
  ------------------
  211|      0|        c->resize = BF(dav1d_resize, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  212|      0|        c->warp8x8  = BF(dav1d_warp_affine_8x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  213|      0|        c->warp8x8t = BF(dav1d_warp_affine_8x8t, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  214|      0|    }
  215|      0|#endif
  216|      0|}

msac.c:msac_init_x86:
   59|  50.5k|static ALWAYS_INLINE void msac_init_x86(MsacContext *const s) {
   60|  50.5k|    const unsigned flags = dav1d_get_cpu_flags();
   61|       |
   62|  50.5k|    if (flags & DAV1D_X86_CPU_FLAG_SSE2) {
  ------------------
  |  Branch (62:9): [True: 50.5k, False: 0]
  ------------------
   63|  50.5k|        s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt16_sse2;
   64|  50.5k|    }
   65|       |
   66|  50.5k|    if (flags & DAV1D_X86_CPU_FLAG_AVX2) {
  ------------------
  |  Branch (66:9): [True: 50.5k, False: 0]
  ------------------
   67|  50.5k|        s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt16_avx2;
   68|  50.5k|    }
   69|  50.5k|}

pal.c:pal_dsp_init_x86:
   34|  10.2k|static ALWAYS_INLINE void pal_dsp_init_x86(Dav1dPalDSPContext *const c) {
   35|  10.2k|    const unsigned flags = dav1d_get_cpu_flags();
   36|       |
   37|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (37:9): [True: 0, False: 10.2k]
  ------------------
   38|       |
   39|  10.2k|    c->pal_idx_finish = dav1d_pal_idx_finish_ssse3;
   40|       |
   41|  10.2k|#if ARCH_X86_64
   42|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (42:9): [True: 0, False: 10.2k]
  ------------------
   43|       |
   44|  10.2k|    c->pal_idx_finish = dav1d_pal_idx_finish_avx2;
   45|       |
   46|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (46:9): [True: 10.2k, False: 0]
  ------------------
   47|       |
   48|      0|    c->pal_idx_finish = dav1d_pal_idx_finish_avx512icl;
   49|      0|#endif
   50|      0|}

refmvs.c:refmvs_dsp_init_x86:
   41|  10.2k|static ALWAYS_INLINE void refmvs_dsp_init_x86(Dav1dRefmvsDSPContext *const c) {
   42|  10.2k|    const unsigned flags = dav1d_get_cpu_flags();
   43|       |
   44|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (44:9): [True: 0, False: 10.2k]
  ------------------
   45|       |
   46|  10.2k|    c->splat_mv = dav1d_splat_mv_sse2;
   47|       |
   48|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (48:9): [True: 0, False: 10.2k]
  ------------------
   49|       |
   50|  10.2k|    c->save_tmvs = dav1d_save_tmvs_ssse3;
   51|       |
   52|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (52:9): [True: 0, False: 10.2k]
  ------------------
   53|  10.2k|#if ARCH_X86_64
   54|  10.2k|    c->load_tmvs = dav1d_load_tmvs_sse4;
   55|       |
   56|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (56:9): [True: 0, False: 10.2k]
  ------------------
   57|       |
   58|  10.2k|    c->save_tmvs = dav1d_save_tmvs_avx2;
   59|  10.2k|    c->splat_mv = dav1d_splat_mv_avx2;
   60|       |
   61|  10.2k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (61:9): [True: 10.2k, False: 0]
  ------------------
   62|       |
   63|      0|    c->save_tmvs = dav1d_save_tmvs_avx512icl;
   64|      0|    c->splat_mv = dav1d_splat_mv_avx512icl;
   65|      0|#endif
   66|      0|}

LLVMFuzzerInitialize:
   59|      2|int LLVMFuzzerInitialize(int *argc, char ***argv) {
   60|      2|    int i = 1;
   61|     11|    for (; i < *argc; i++) {
  ------------------
  |  Branch (61:12): [True: 9, False: 2]
  ------------------
   62|      9|        if (!strcmp((*argv)[i], "--cpumask")) {
  ------------------
  |  Branch (62:13): [True: 0, False: 9]
  ------------------
   63|      0|            const char * cpumask = (*argv)[i+1];
   64|      0|            if (cpumask) {
  ------------------
  |  Branch (64:17): [True: 0, False: 0]
  ------------------
   65|      0|                char *end;
   66|      0|                unsigned res;
   67|      0|                if (!strncmp(cpumask, "0x", 2)) {
  ------------------
  |  Branch (67:21): [True: 0, False: 0]
  ------------------
   68|      0|                    cpumask += 2;
   69|      0|                    res = (unsigned) strtoul(cpumask, &end, 16);
   70|      0|                } else {
   71|      0|                    res = (unsigned) strtoul(cpumask, &end, 0);
   72|      0|                }
   73|      0|                if (end != cpumask && !end[0]) {
  ------------------
  |  Branch (73:21): [True: 0, False: 0]
  |  Branch (73:39): [True: 0, False: 0]
  ------------------
   74|      0|                    dav1d_set_cpu_flags_mask(res);
   75|      0|                }
   76|      0|            }
   77|      0|            break;
   78|      0|        }
   79|      9|    }
   80|       |
   81|      2|    for (; i < *argc - 2; i++) {
  ------------------
  |  Branch (81:12): [True: 0, False: 2]
  ------------------
   82|      0|        (*argv)[i] = (*argv)[i + 2];
   83|      0|    }
   84|       |
   85|      2|    *argc = i;
   86|       |
   87|      2|    return 0;
   88|      2|}
LLVMFuzzerTestOneInput:
   94|  10.2k|{
   95|  10.2k|    Dav1dSettings settings = { 0 };
   96|  10.2k|    Dav1dContext * ctx = NULL;
   97|  10.2k|    Dav1dPicture pic;
   98|  10.2k|    const uint8_t *ptr = data;
   99|  10.2k|    int have_seq_hdr = 0;
  100|  10.2k|    int err;
  101|       |
  102|  10.2k|    dav1d_version();
  103|       |
  104|  10.2k|    if (size < 32) goto end;
  ------------------
  |  Branch (104:9): [True: 8, False: 10.2k]
  ------------------
  105|       |#ifdef DAV1D_ALLOC_FAIL
  106|       |    unsigned h = djb_xor(ptr, 32);
  107|       |    unsigned seed = h;
  108|       |    unsigned probability = h > (RAND_MAX >> 5) ? RAND_MAX >> 5 : h;
  109|       |    int max_frame_delay = (h & 0xf) + 1;
  110|       |    int n_threads = ((h >> 4) & 0x7) + 1;
  111|       |    if (max_frame_delay > 5) max_frame_delay = 1;
  112|       |    if (n_threads > 3) n_threads = 1;
  113|       |#endif
  114|  10.2k|    ptr += 32; // skip ivf header
  115|       |
  116|  10.2k|    dav1d_default_settings(&settings);
  117|       |
  118|       |#ifdef DAV1D_MT_FUZZING
  119|       |    settings.max_frame_delay = settings.n_threads = 4;
  120|       |#elif defined(DAV1D_ALLOC_FAIL)
  121|       |    settings.max_frame_delay = max_frame_delay;
  122|       |    settings.n_threads = n_threads;
  123|       |    dav1d_setup_alloc_fail(seed, probability);
  124|       |#else
  125|  10.2k|    settings.max_frame_delay = settings.n_threads = 1;
  126|  10.2k|#endif
  127|  10.2k|#if defined(DAV1D_FUZZ_MAX_SIZE)
  128|  10.2k|    settings.frame_size_limit = DAV1D_FUZZ_MAX_SIZE;
  ------------------
  |  |   56|  10.2k|#define DAV1D_FUZZ_MAX_SIZE 4096 * 4096
  ------------------
  129|  10.2k|#endif
  130|       |
  131|  10.2k|    err = dav1d_open(&ctx, &settings);
  132|  10.2k|    if (err < 0) goto end;
  ------------------
  |  Branch (132:9): [True: 0, False: 10.2k]
  ------------------
  133|       |
  134|  93.5k|    while (ptr <= data + size - 12) {
  ------------------
  |  Branch (134:12): [True: 83.8k, False: 9.72k]
  ------------------
  135|  83.8k|        Dav1dData buf;
  136|  83.8k|        uint8_t *p;
  137|       |
  138|  83.8k|        size_t frame_size = r32le(ptr);
  139|  83.8k|        ptr += 12;
  140|       |
  141|  83.8k|        if (frame_size > size || ptr > data + size - frame_size)
  ------------------
  |  Branch (141:13): [True: 355, False: 83.4k]
  |  Branch (141:34): [True: 149, False: 83.3k]
  ------------------
  142|    504|            break;
  143|       |
  144|  83.3k|        if (!frame_size) continue;
  ------------------
  |  Branch (144:13): [True: 1.05k, False: 82.2k]
  ------------------
  145|       |
  146|  82.2k|        if (!have_seq_hdr) {
  ------------------
  |  Branch (146:13): [True: 13.1k, False: 69.0k]
  ------------------
  147|  13.1k|            Dav1dSequenceHeader seq;
  148|  13.1k|            int err = dav1d_parse_sequence_header(&seq, ptr, frame_size);
  149|       |            // skip frames until we see a sequence header
  150|  13.1k|            if  (err != 0) {
  ------------------
  |  Branch (150:18): [True: 3.34k, False: 9.84k]
  ------------------
  151|  3.34k|                ptr += frame_size;
  152|  3.34k|                continue;
  153|  3.34k|            }
  154|  9.84k|            have_seq_hdr = 1;
  155|  9.84k|        }
  156|       |
  157|       |        // copy frame data to a new buffer to catch reads past the end of input
  158|  78.9k|        p = dav1d_data_create(&buf, frame_size);
  159|  78.9k|        if (!p) goto cleanup;
  ------------------
  |  Branch (159:13): [True: 0, False: 78.9k]
  ------------------
  160|  78.9k|        memcpy(p, ptr, frame_size);
  161|  78.9k|        ptr += frame_size;
  162|       |
  163|  82.2k|        do {
  164|  82.2k|            if ((err = dav1d_send_data(ctx, &buf)) < 0) {
  ------------------
  |  Branch (164:17): [True: 57.0k, False: 25.1k]
  ------------------
  165|  57.0k|                if (err != DAV1D_ERR(EAGAIN))
  ------------------
  |  |   58|  57.0k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (165:21): [True: 53.6k, False: 3.43k]
  ------------------
  166|  53.6k|                    break;
  167|  57.0k|            }
  168|  28.6k|            memset(&pic, 0, sizeof(pic));
  169|  28.6k|            err = dav1d_get_picture(ctx, &pic);
  170|  28.6k|            if (err == 0) {
  ------------------
  |  Branch (170:17): [True: 21.1k, False: 7.41k]
  ------------------
  171|  21.1k|                dav1d_picture_unref(&pic);
  172|  21.1k|            } else if (err != DAV1D_ERR(EAGAIN)) {
  ------------------
  |  |   58|  7.41k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (172:24): [True: 115, False: 7.30k]
  ------------------
  173|    115|                break;
  174|    115|            }
  175|  28.6k|        } while (buf.sz > 0);
  ------------------
  |  Branch (175:18): [True: 3.32k, False: 25.1k]
  ------------------
  176|       |
  177|  78.9k|        if (buf.sz > 0)
  ------------------
  |  Branch (177:13): [True: 53.7k, False: 25.1k]
  ------------------
  178|  53.7k|            dav1d_data_unref(&buf);
  179|  78.9k|    }
  180|       |
  181|  10.2k|    memset(&pic, 0, sizeof(pic));
  182|  10.2k|    if ((err = dav1d_get_picture(ctx, &pic)) == 0) {
  ------------------
  |  Branch (182:9): [True: 251, False: 9.97k]
  ------------------
  183|       |        /* Test calling dav1d_picture_unref() after dav1d_close() */
  184|  1.21k|        do {
  185|  1.21k|            Dav1dPicture pic2 = { 0 };
  186|  1.21k|            if ((err = dav1d_get_picture(ctx, &pic2)) == 0)
  ------------------
  |  Branch (186:17): [True: 826, False: 390]
  ------------------
  187|    826|                dav1d_picture_unref(&pic2);
  188|  1.21k|        } while (err != DAV1D_ERR(EAGAIN));
  ------------------
  |  |   58|  1.21k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (188:18): [True: 965, False: 251]
  ------------------
  189|       |
  190|    251|        dav1d_close(&ctx);
  191|    251|        dav1d_picture_unref(&pic);
  192|    251|        return 0;
  193|    251|    }
  194|       |
  195|  9.97k|cleanup:
  196|  9.97k|    dav1d_close(&ctx);
  197|  9.98k|end:
  198|  9.98k|    return 0;
  199|  9.97k|}
dav1d_fuzzer.c:r32le:
   52|  83.8k|static unsigned r32le(const uint8_t *const p) {
   53|  83.8k|    return ((uint32_t)p[3] << 24U) | (p[2] << 16U) | (p[1] << 8U) | p[0];
   54|  83.8k|}

