CBS_init:
   29|  11.4k|void CBS_init(CBS *cbs, const uint8_t *data, size_t len) {
   30|  11.4k|  cbs->data = data;
   31|  11.4k|  cbs->len = len;
   32|  11.4k|}
CBS_data:
   50|  8.36k|const uint8_t *CBS_data(const CBS *cbs) {
   51|  8.36k|  return cbs->data;
   52|  8.36k|}
CBS_len:
   54|  25.2k|size_t CBS_len(const CBS *cbs) {
   55|  25.2k|  return cbs->len;
   56|  25.2k|}
CBS_get_u8:
  108|  2.87k|int CBS_get_u8(CBS *cbs, uint8_t *out) {
  109|  2.87k|  const uint8_t *v;
  110|  2.87k|  if (!cbs_get(cbs, &v, 1)) {
  ------------------
  |  Branch (110:7): [True: 20, False: 2.85k]
  ------------------
  111|     20|    return 0;
  112|     20|  }
  113|  2.85k|  *out = *v;
  114|  2.85k|  return 1;
  115|  2.87k|}
CBS_get_bytes:
  181|  8.54k|int CBS_get_bytes(CBS *cbs, CBS *out, size_t len) {
  182|  8.54k|  const uint8_t *v;
  183|  8.54k|  if (!cbs_get(cbs, &v, len)) {
  ------------------
  |  Branch (183:7): [True: 22, False: 8.52k]
  ------------------
  184|     22|    return 0;
  185|     22|  }
  186|  8.52k|  CBS_init(out, v, len);
  187|  8.52k|  return 1;
  188|  8.54k|}
CBS_get_u16_length_prefixed:
  214|  8.58k|int CBS_get_u16_length_prefixed(CBS *cbs, CBS *out) {
  215|  8.58k|  return cbs_get_length_prefixed(cbs, out, 2);
  216|  8.58k|}
cbs.c:cbs_get:
   34|  20.0k|static int cbs_get(CBS *cbs, const uint8_t **p, size_t n) {
   35|  20.0k|  if (cbs->len < n) {
  ------------------
  |  Branch (35:7): [True: 80, False: 19.9k]
  ------------------
   36|     80|    return 0;
   37|     80|  }
   38|       |
   39|  19.9k|  *p = cbs->data;
   40|  19.9k|  cbs->data += n;
   41|  19.9k|  cbs->len -= n;
   42|  19.9k|  return 1;
   43|  20.0k|}
cbs.c:cbs_get_u:
   93|  8.58k|static int cbs_get_u(CBS *cbs, uint64_t *out, size_t len) {
   94|  8.58k|  uint64_t result = 0;
   95|  8.58k|  const uint8_t *data;
   96|       |
   97|  8.58k|  if (!cbs_get(cbs, &data, len)) {
  ------------------
  |  Branch (97:7): [True: 38, False: 8.54k]
  ------------------
   98|     38|    return 0;
   99|     38|  }
  100|  25.6k|  for (size_t i = 0; i < len; i++) {
  ------------------
  |  Branch (100:22): [True: 17.0k, False: 8.54k]
  ------------------
  101|  17.0k|    result <<= 8;
  102|  17.0k|    result |= data[i];
  103|  17.0k|  }
  104|  8.54k|  *out = result;
  105|  8.54k|  return 1;
  106|  8.58k|}
cbs.c:cbs_get_length_prefixed:
  199|  8.58k|static int cbs_get_length_prefixed(CBS *cbs, CBS *out, size_t len_len) {
  200|  8.58k|  uint64_t len;
  201|  8.58k|  if (!cbs_get_u(cbs, &len, len_len)) {
  ------------------
  |  Branch (201:7): [True: 38, False: 8.54k]
  ------------------
  202|     38|    return 0;
  203|     38|  }
  204|       |  // If |len_len| <= 3 then we know that |len| will fit into a |size_t|, even on
  205|       |  // 32-bit systems.
  206|  8.54k|  assert(len_len <= 3);
  207|  8.54k|  return CBS_get_bytes(cbs, out, len);
  208|  8.54k|}

OPENSSL_cpuid_setup:
  153|      2|void OPENSSL_cpuid_setup(void) {
  154|       |  // Determine the vendor and maximum input value.
  155|      2|  uint32_t eax, ebx, ecx, edx;
  156|      2|  OPENSSL_cpuid(&eax, &ebx, &ecx, &edx, 0);
  157|       |
  158|      2|  uint32_t num_ids = eax;
  159|       |
  160|      2|  int is_intel = ebx == 0x756e6547 /* Genu */ &&
  ------------------
  |  Branch (160:18): [True: 2, False: 0]
  ------------------
  161|      2|                 edx == 0x49656e69 /* ineI */ &&
  ------------------
  |  Branch (161:18): [True: 2, False: 0]
  ------------------
  162|      2|                 ecx == 0x6c65746e /* ntel */;
  ------------------
  |  Branch (162:18): [True: 2, False: 0]
  ------------------
  163|      2|  int is_amd = ebx == 0x68747541 /* Auth */ &&
  ------------------
  |  Branch (163:16): [True: 0, False: 2]
  ------------------
  164|      2|               edx == 0x69746e65 /* enti */ &&
  ------------------
  |  Branch (164:16): [True: 0, False: 0]
  ------------------
  165|      2|               ecx == 0x444d4163 /* cAMD */;
  ------------------
  |  Branch (165:16): [True: 0, False: 0]
  ------------------
  166|       |
  167|      2|  uint32_t extended_features[2] = {0};
  168|      2|  if (num_ids >= 7) {
  ------------------
  |  Branch (168:7): [True: 2, False: 0]
  ------------------
  169|      2|    OPENSSL_cpuid(&eax, &ebx, &ecx, &edx, 7);
  170|      2|    extended_features[0] = ebx;
  171|      2|    extended_features[1] = ecx;
  172|      2|  }
  173|       |
  174|      2|  OPENSSL_cpuid(&eax, &ebx, &ecx, &edx, 1);
  175|       |
  176|      2|  if (is_amd) {
  ------------------
  |  Branch (176:7): [True: 0, False: 2]
  ------------------
  177|       |    // See https://www.amd.com/system/files/TechDocs/25481.pdf, page 10.
  178|      0|    const uint32_t base_family = (eax >> 8) & 15;
  179|      0|    const uint32_t base_model = (eax >> 4) & 15;
  180|       |
  181|      0|    uint32_t family = base_family;
  182|      0|    uint32_t model = base_model;
  183|      0|    if (base_family == 0xf) {
  ------------------
  |  Branch (183:9): [True: 0, False: 0]
  ------------------
  184|      0|      const uint32_t ext_family = (eax >> 20) & 255;
  185|      0|      family += ext_family;
  186|      0|      const uint32_t ext_model = (eax >> 16) & 15;
  187|      0|      model |= ext_model << 4;
  188|      0|    }
  189|       |
  190|      0|    if (family < 0x17 || (family == 0x17 && 0x70 <= model && model <= 0x7f)) {
  ------------------
  |  Branch (190:9): [True: 0, False: 0]
  |  Branch (190:27): [True: 0, False: 0]
  |  Branch (190:45): [True: 0, False: 0]
  |  Branch (190:62): [True: 0, False: 0]
  ------------------
  191|       |      // Disable RDRAND on AMD families before 0x17 (Zen) due to reported
  192|       |      // failures after suspend.
  193|       |      // https://bugzilla.redhat.com/show_bug.cgi?id=1150286
  194|       |      // Also disable for family 0x17, models 0x70–0x7f, due to possible RDRAND
  195|       |      // failures there too.
  196|      0|      ecx &= ~(1u << 30);
  197|      0|    }
  198|      0|  }
  199|       |
  200|       |  // Force the hyper-threading bit so that the more conservative path is always
  201|       |  // chosen.
  202|      2|  edx |= 1u << 28;
  203|       |
  204|       |  // Reserved bit #20 was historically repurposed to control the in-memory
  205|       |  // representation of RC4 state. Always set it to zero.
  206|      2|  edx &= ~(1u << 20);
  207|       |
  208|       |  // Reserved bit #30 is repurposed to signal an Intel CPU.
  209|      2|  if (is_intel) {
  ------------------
  |  Branch (209:7): [True: 2, False: 0]
  ------------------
  210|      2|    edx |= (1u << 30);
  211|       |
  212|       |    // Clear the XSAVE bit on Knights Landing to mimic Silvermont. This enables
  213|       |    // some Silvermont-specific codepaths which perform better. See OpenSSL
  214|       |    // commit 64d92d74985ebb3d0be58a9718f9e080a14a8e7f.
  215|      2|    if ((eax & 0x0fff0ff0) == 0x00050670 /* Knights Landing */ ||
  ------------------
  |  Branch (215:9): [True: 0, False: 2]
  ------------------
  216|      2|        (eax & 0x0fff0ff0) == 0x00080650 /* Knights Mill (per SDE) */) {
  ------------------
  |  Branch (216:9): [True: 0, False: 2]
  ------------------
  217|      0|      ecx &= ~(1u << 26);
  218|      0|    }
  219|      2|  } else {
  220|      0|    edx &= ~(1u << 30);
  221|      0|  }
  222|       |
  223|       |  // The SDBG bit is repurposed to denote AMD XOP support. Don't ever use AMD
  224|       |  // XOP code paths.
  225|      2|  ecx &= ~(1u << 11);
  226|       |
  227|      2|  uint64_t xcr0 = 0;
  228|      2|  if (ecx & (1u << 27)) {
  ------------------
  |  Branch (228:7): [True: 2, False: 0]
  ------------------
  229|       |    // XCR0 may only be queried if the OSXSAVE bit is set.
  230|      2|    xcr0 = OPENSSL_xgetbv(0);
  231|      2|  }
  232|       |  // See Intel manual, volume 1, section 14.3.
  233|      2|  if ((xcr0 & 6) != 6) {
  ------------------
  |  Branch (233:7): [True: 0, False: 2]
  ------------------
  234|       |    // YMM registers cannot be used.
  235|      0|    ecx &= ~(1u << 28);  // AVX
  236|      0|    ecx &= ~(1u << 12);  // FMA
  237|      0|    ecx &= ~(1u << 11);  // AMD XOP
  238|       |    // Clear AVX2 and AVX512* bits.
  239|       |    //
  240|       |    // TODO(davidben): Should bits 17 and 26-28 also be cleared? Upstream
  241|       |    // doesn't clear those.
  242|      0|    extended_features[0] &=
  243|      0|        ~((1u << 5) | (1u << 16) | (1u << 21) | (1u << 30) | (1u << 31));
  244|      0|  }
  245|       |  // See Intel manual, volume 1, section 15.2.
  246|      2|  if ((xcr0 & 0xe6) != 0xe6) {
  ------------------
  |  Branch (246:7): [True: 2, False: 0]
  ------------------
  247|       |    // Clear AVX512F. Note we don't touch other AVX512 extensions because they
  248|       |    // can be used with YMM.
  249|      2|    extended_features[0] &= ~(1u << 16);
  250|      2|  }
  251|       |
  252|       |  // Disable ADX instructions on Knights Landing. See OpenSSL commit
  253|       |  // 64d92d74985ebb3d0be58a9718f9e080a14a8e7f.
  254|      2|  if ((ecx & (1u << 26)) == 0) {
  ------------------
  |  Branch (254:7): [True: 0, False: 2]
  ------------------
  255|      0|    extended_features[0] &= ~(1u << 19);
  256|      0|  }
  257|       |
  258|      2|  OPENSSL_ia32cap_P[0] = edx;
  259|      2|  OPENSSL_ia32cap_P[1] = ecx;
  260|      2|  OPENSSL_ia32cap_P[2] = extended_features[0];
  261|      2|  OPENSSL_ia32cap_P[3] = extended_features[1];
  262|       |
  263|      2|  const char *env1, *env2;
  264|      2|  env1 = getenv("OPENSSL_ia32cap");
  265|      2|  if (env1 == NULL) {
  ------------------
  |  Branch (265:7): [True: 2, False: 0]
  ------------------
  266|      2|    return;
  267|      2|  }
  268|       |
  269|       |  // OPENSSL_ia32cap can contain zero, one or two values, separated with a ':'.
  270|       |  // Each value is a 64-bit, unsigned value which may start with "0x" to
  271|       |  // indicate a hex value. Prior to the 64-bit value, a '~' or '|' may be given.
  272|       |  //
  273|       |  // If the '~' prefix is present:
  274|       |  //   the value is inverted and ANDed with the probed CPUID result
  275|       |  // If the '|' prefix is present:
  276|       |  //   the value is ORed with the probed CPUID result
  277|       |  // Otherwise:
  278|       |  //   the value is taken as the result of the CPUID
  279|       |  //
  280|       |  // The first value determines OPENSSL_ia32cap_P[0] and [1]. The second [2]
  281|       |  // and [3].
  282|       |
  283|      0|  handle_cpu_env(&OPENSSL_ia32cap_P[0], env1);
  284|      0|  env2 = strchr(env1, ':');
  285|      0|  if (env2 != NULL) {
  ------------------
  |  Branch (285:7): [True: 0, False: 0]
  ------------------
  286|      0|    handle_cpu_env(&OPENSSL_ia32cap_P[2], env2 + 1);
  287|      0|  }
  288|      0|}
cpu_intel.c:OPENSSL_cpuid:
   80|      6|                          uint32_t *out_ecx, uint32_t *out_edx, uint32_t leaf) {
   81|       |#if defined(_MSC_VER)
   82|       |  int tmp[4];
   83|       |  __cpuid(tmp, (int)leaf);
   84|       |  *out_eax = (uint32_t)tmp[0];
   85|       |  *out_ebx = (uint32_t)tmp[1];
   86|       |  *out_ecx = (uint32_t)tmp[2];
   87|       |  *out_edx = (uint32_t)tmp[3];
   88|       |#elif defined(__pic__) && defined(OPENSSL_32_BIT)
   89|       |  // Inline assembly may not clobber the PIC register. For 32-bit, this is EBX.
   90|       |  // See https://gcc.gnu.org/bugzilla/show_bug.cgi?id=47602.
   91|       |  __asm__ volatile (
   92|       |    "xor %%ecx, %%ecx\n"
   93|       |    "mov %%ebx, %%edi\n"
   94|       |    "cpuid\n"
   95|       |    "xchg %%edi, %%ebx\n"
   96|       |    : "=a"(*out_eax), "=D"(*out_ebx), "=c"(*out_ecx), "=d"(*out_edx)
   97|       |    : "a"(leaf)
   98|       |  );
   99|       |#else
  100|      6|  __asm__ volatile (
  101|      6|    "xor %%ecx, %%ecx\n"
  102|      6|    "cpuid\n"
  103|      6|    : "=a"(*out_eax), "=b"(*out_ebx), "=c"(*out_ecx), "=d"(*out_edx)
  104|      6|    : "a"(leaf)
  105|      6|  );
  106|      6|#endif
  107|      6|}
cpu_intel.c:OPENSSL_xgetbv:
  111|      2|static uint64_t OPENSSL_xgetbv(uint32_t xcr) {
  112|       |#if defined(_MSC_VER)
  113|       |  return (uint64_t)_xgetbv(xcr);
  114|       |#else
  115|      2|  uint32_t eax, edx;
  116|      2|  __asm__ volatile ("xgetbv" : "=a"(eax), "=d"(edx) : "c"(xcr));
  117|      2|  return (((uint64_t)edx) << 32) | eax;
  118|      2|#endif
  119|      2|}

crypto.c:do_library_init:
  151|      2|static void OPENSSL_CDECL do_library_init(void) {
  152|       | // WARNING: this function may only configure the capability variables. See the
  153|       | // note above about the linker bug.
  154|      2|#if defined(NEED_CPUID)
  155|      2|  OPENSSL_cpuid_setup();
  156|      2|#endif
  157|      2|}

BN_add:
   67|  4.65k|int BN_add(BIGNUM *r, const BIGNUM *a, const BIGNUM *b) {
   68|  4.65k|  const BIGNUM *tmp;
   69|  4.65k|  int a_neg = a->neg, ret;
   70|       |
   71|       |  //  a +  b	a+b
   72|       |  //  a + -b	a-b
   73|       |  // -a +  b	b-a
   74|       |  // -a + -b	-(a+b)
   75|  4.65k|  if (a_neg ^ b->neg) {
  ------------------
  |  Branch (75:7): [True: 4.65k, False: 0]
  ------------------
   76|       |    // only one is negative
   77|  4.65k|    if (a_neg) {
  ------------------
  |  Branch (77:9): [True: 4.65k, False: 0]
  ------------------
   78|  4.65k|      tmp = a;
   79|  4.65k|      a = b;
   80|  4.65k|      b = tmp;
   81|  4.65k|    }
   82|       |
   83|       |    // we are now a - b
   84|  4.65k|    if (BN_ucmp(a, b) < 0) {
  ------------------
  |  Branch (84:9): [True: 0, False: 4.65k]
  ------------------
   85|      0|      if (!BN_usub(r, b, a)) {
  ------------------
  |  Branch (85:11): [True: 0, False: 0]
  ------------------
   86|      0|        return 0;
   87|      0|      }
   88|      0|      r->neg = 1;
   89|  4.65k|    } else {
   90|  4.65k|      if (!BN_usub(r, a, b)) {
  ------------------
  |  Branch (90:11): [True: 0, False: 4.65k]
  ------------------
   91|      0|        return 0;
   92|      0|      }
   93|  4.65k|      r->neg = 0;
   94|  4.65k|    }
   95|  4.65k|    return 1;
   96|  4.65k|  }
   97|       |
   98|      0|  ret = BN_uadd(r, a, b);
   99|      0|  r->neg = a_neg;
  100|      0|  return ret;
  101|  4.65k|}
BN_add_word:
  138|  54.5k|int BN_add_word(BIGNUM *a, BN_ULONG w) {
  139|  54.5k|  BN_ULONG l;
  140|  54.5k|  int i;
  141|       |
  142|       |  // degenerate case: w is zero
  143|  54.5k|  if (!w) {
  ------------------
  |  Branch (143:7): [True: 0, False: 54.5k]
  ------------------
  144|      0|    return 1;
  145|      0|  }
  146|       |
  147|       |  // degenerate case: a is zero
  148|  54.5k|  if (BN_is_zero(a)) {
  ------------------
  |  Branch (148:7): [True: 5.20k, False: 49.3k]
  ------------------
  149|  5.20k|    return BN_set_word(a, w);
  150|  5.20k|  }
  151|       |
  152|       |  // handle 'a' when negative
  153|  49.3k|  if (a->neg) {
  ------------------
  |  Branch (153:7): [True: 0, False: 49.3k]
  ------------------
  154|      0|    a->neg = 0;
  155|      0|    i = BN_sub_word(a, w);
  156|      0|    if (!BN_is_zero(a)) {
  ------------------
  |  Branch (156:9): [True: 0, False: 0]
  ------------------
  157|      0|      a->neg = !(a->neg);
  158|      0|    }
  159|      0|    return i;
  160|      0|  }
  161|       |
  162|  98.9k|  for (i = 0; w != 0 && i < a->width; i++) {
  ------------------
  |  Branch (162:15): [True: 49.8k, False: 49.1k]
  |  Branch (162:25): [True: 49.6k, False: 204]
  ------------------
  163|  49.6k|    a->d[i] = l = a->d[i] + w;
  164|  49.6k|    w = (w > l) ? 1 : 0;
  ------------------
  |  Branch (164:9): [True: 461, False: 49.1k]
  ------------------
  165|  49.6k|  }
  166|       |
  167|  49.3k|  if (w && i == a->width) {
  ------------------
  |  Branch (167:7): [True: 204, False: 49.1k]
  |  Branch (167:12): [True: 204, False: 0]
  ------------------
  168|    204|    if (!bn_wexpand(a, a->width + 1)) {
  ------------------
  |  Branch (168:9): [True: 0, False: 204]
  ------------------
  169|      0|      return 0;
  170|      0|    }
  171|    204|    a->width++;
  172|    204|    a->d[i] = w;
  173|    204|  }
  174|       |
  175|  49.3k|  return 1;
  176|  49.3k|}
bn_usub_consttime:
  226|   137k|int bn_usub_consttime(BIGNUM *r, const BIGNUM *a, const BIGNUM *b) {
  227|       |  // |b| may have more words than |a| given non-minimal inputs, but all words
  228|       |  // beyond |a->width| must then be zero.
  229|   137k|  int b_width = b->width;
  230|   137k|  if (b_width > a->width) {
  ------------------
  |  Branch (230:7): [True: 3.51k, False: 134k]
  ------------------
  231|  3.51k|    if (!bn_fits_in_words(b, a->width)) {
  ------------------
  |  Branch (231:9): [True: 0, False: 3.51k]
  ------------------
  232|      0|      OPENSSL_PUT_ERROR(BN, BN_R_ARG2_LT_ARG3);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  233|      0|      return 0;
  234|      0|    }
  235|  3.51k|    b_width = a->width;
  236|  3.51k|  }
  237|       |
  238|   137k|  if (!bn_wexpand(r, a->width)) {
  ------------------
  |  Branch (238:7): [True: 0, False: 137k]
  ------------------
  239|      0|    return 0;
  240|      0|  }
  241|       |
  242|   137k|  BN_ULONG borrow = bn_sub_words(r->d, a->d, b->d, b_width);
  243|   176k|  for (int i = b_width; i < a->width; i++) {
  ------------------
  |  Branch (243:25): [True: 39.2k, False: 137k]
  ------------------
  244|       |    // |r| and |a| may alias, so use a temporary.
  245|  39.2k|    BN_ULONG tmp = a->d[i];
  246|  39.2k|    r->d[i] = a->d[i] - borrow;
  247|  39.2k|    borrow = tmp < r->d[i];
  248|  39.2k|  }
  249|       |
  250|   137k|  if (borrow) {
  ------------------
  |  Branch (250:7): [True: 0, False: 137k]
  ------------------
  251|      0|    OPENSSL_PUT_ERROR(BN, BN_R_ARG2_LT_ARG3);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  252|      0|    return 0;
  253|      0|  }
  254|       |
  255|   137k|  r->width = a->width;
  256|   137k|  r->neg = 0;
  257|   137k|  return 1;
  258|   137k|}
BN_usub:
  260|   137k|int BN_usub(BIGNUM *r, const BIGNUM *a, const BIGNUM *b) {
  261|   137k|  if (!bn_usub_consttime(r, a, b)) {
  ------------------
  |  Branch (261:7): [True: 0, False: 137k]
  ------------------
  262|      0|    return 0;
  263|      0|  }
  264|   137k|  bn_set_minimal_width(r);
  265|   137k|  return 1;
  266|   137k|}

bn_mul_add_words:
   98|   810k|                          BN_ULONG w) {
   99|   810k|  BN_ULONG c1 = 0;
  100|       |
  101|   810k|  if (num == 0) {
  ------------------
  |  Branch (101:7): [True: 0, False: 810k]
  ------------------
  102|      0|    return (c1);
  103|      0|  }
  104|       |
  105|  2.26M|  while (num & ~3) {
  ------------------
  |  Branch (105:10): [True: 1.45M, False: 810k]
  ------------------
  106|  1.45M|    mul_add(rp[0], ap[0], w, c1);
  ------------------
  |  |   69|  1.45M|  do {                                                                     \
  |  |   70|  1.45M|    register BN_ULONG high, low;                                           \
  |  |   71|  1.45M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|  1.45M|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|  1.45M|            : "a"(low), "g"(0)                                             \
  |  |   75|  1.45M|            : "cc");                                                       \
  |  |   76|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|  1.45M|            : "+m"(r), "+d"(high)                                          \
  |  |   78|  1.45M|            : "r"(carry), "g"(0)                                           \
  |  |   79|  1.45M|            : "cc");                                                       \
  |  |   80|  1.45M|    (carry) = high;                                                        \
  |  |   81|  1.45M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  107|  1.45M|    mul_add(rp[1], ap[1], w, c1);
  ------------------
  |  |   69|  1.45M|  do {                                                                     \
  |  |   70|  1.45M|    register BN_ULONG high, low;                                           \
  |  |   71|  1.45M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|  1.45M|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|  1.45M|            : "a"(low), "g"(0)                                             \
  |  |   75|  1.45M|            : "cc");                                                       \
  |  |   76|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|  1.45M|            : "+m"(r), "+d"(high)                                          \
  |  |   78|  1.45M|            : "r"(carry), "g"(0)                                           \
  |  |   79|  1.45M|            : "cc");                                                       \
  |  |   80|  1.45M|    (carry) = high;                                                        \
  |  |   81|  1.45M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  108|  1.45M|    mul_add(rp[2], ap[2], w, c1);
  ------------------
  |  |   69|  1.45M|  do {                                                                     \
  |  |   70|  1.45M|    register BN_ULONG high, low;                                           \
  |  |   71|  1.45M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|  1.45M|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|  1.45M|            : "a"(low), "g"(0)                                             \
  |  |   75|  1.45M|            : "cc");                                                       \
  |  |   76|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|  1.45M|            : "+m"(r), "+d"(high)                                          \
  |  |   78|  1.45M|            : "r"(carry), "g"(0)                                           \
  |  |   79|  1.45M|            : "cc");                                                       \
  |  |   80|  1.45M|    (carry) = high;                                                        \
  |  |   81|  1.45M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  109|  1.45M|    mul_add(rp[3], ap[3], w, c1);
  ------------------
  |  |   69|  1.45M|  do {                                                                     \
  |  |   70|  1.45M|    register BN_ULONG high, low;                                           \
  |  |   71|  1.45M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|  1.45M|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|  1.45M|            : "a"(low), "g"(0)                                             \
  |  |   75|  1.45M|            : "cc");                                                       \
  |  |   76|  1.45M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|  1.45M|            : "+m"(r), "+d"(high)                                          \
  |  |   78|  1.45M|            : "r"(carry), "g"(0)                                           \
  |  |   79|  1.45M|            : "cc");                                                       \
  |  |   80|  1.45M|    (carry) = high;                                                        \
  |  |   81|  1.45M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  110|  1.45M|    ap += 4;
  111|  1.45M|    rp += 4;
  112|  1.45M|    num -= 4;
  113|  1.45M|  }
  114|   810k|  if (num) {
  ------------------
  |  Branch (114:7): [True: 658k, False: 151k]
  ------------------
  115|   658k|    mul_add(rp[0], ap[0], w, c1);
  ------------------
  |  |   69|   658k|  do {                                                                     \
  |  |   70|   658k|    register BN_ULONG high, low;                                           \
  |  |   71|   658k|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|   658k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|   658k|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|   658k|            : "a"(low), "g"(0)                                             \
  |  |   75|   658k|            : "cc");                                                       \
  |  |   76|   658k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|   658k|            : "+m"(r), "+d"(high)                                          \
  |  |   78|   658k|            : "r"(carry), "g"(0)                                           \
  |  |   79|   658k|            : "cc");                                                       \
  |  |   80|   658k|    (carry) = high;                                                        \
  |  |   81|   658k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  116|   658k|    if (--num == 0) {
  ------------------
  |  Branch (116:9): [True: 439k, False: 219k]
  ------------------
  117|   439k|      return c1;
  118|   439k|    }
  119|   219k|    mul_add(rp[1], ap[1], w, c1);
  ------------------
  |  |   69|   219k|  do {                                                                     \
  |  |   70|   219k|    register BN_ULONG high, low;                                           \
  |  |   71|   219k|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|   219k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|   219k|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|   219k|            : "a"(low), "g"(0)                                             \
  |  |   75|   219k|            : "cc");                                                       \
  |  |   76|   219k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|   219k|            : "+m"(r), "+d"(high)                                          \
  |  |   78|   219k|            : "r"(carry), "g"(0)                                           \
  |  |   79|   219k|            : "cc");                                                       \
  |  |   80|   219k|    (carry) = high;                                                        \
  |  |   81|   219k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  120|   219k|    if (--num == 0) {
  ------------------
  |  Branch (120:9): [True: 92.3k, False: 127k]
  ------------------
  121|  92.3k|      return c1;
  122|  92.3k|    }
  123|   127k|    mul_add(rp[2], ap[2], w, c1);
  ------------------
  |  |   69|   127k|  do {                                                                     \
  |  |   70|   127k|    register BN_ULONG high, low;                                           \
  |  |   71|   127k|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "m"(a) : "cc"); \
  |  |   72|   127k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   73|   127k|            : "+r"(carry), "+d"(high)                                      \
  |  |   74|   127k|            : "a"(low), "g"(0)                                             \
  |  |   75|   127k|            : "cc");                                                       \
  |  |   76|   127k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   77|   127k|            : "+m"(r), "+d"(high)                                          \
  |  |   78|   127k|            : "r"(carry), "g"(0)                                           \
  |  |   79|   127k|            : "cc");                                                       \
  |  |   80|   127k|    (carry) = high;                                                        \
  |  |   81|   127k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (81:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  124|   127k|    return c1;
  125|   219k|  }
  126|       |
  127|   151k|  return c1;
  128|   810k|}
bn_mul_words:
  131|  6.94M|                      BN_ULONG w) {
  132|  6.94M|  BN_ULONG c1 = 0;
  133|       |
  134|  6.94M|  if (num == 0) {
  ------------------
  |  Branch (134:7): [True: 0, False: 6.94M]
  ------------------
  135|      0|    return c1;
  136|      0|  }
  137|       |
  138|  37.1M|  while (num & ~3) {
  ------------------
  |  Branch (138:10): [True: 30.1M, False: 6.94M]
  ------------------
  139|  30.1M|    mul(rp[0], ap[0], w, c1);
  ------------------
  |  |   84|  30.1M|  do {                                                                     \
  |  |   85|  30.1M|    register BN_ULONG high, low;                                           \
  |  |   86|  30.1M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  30.1M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  30.1M|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  30.1M|            : "a"(low), "g"(0)                                             \
  |  |   90|  30.1M|            : "cc");                                                       \
  |  |   91|  30.1M|    (r) = (carry);                                                         \
  |  |   92|  30.1M|    (carry) = high;                                                        \
  |  |   93|  30.1M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  140|  30.1M|    mul(rp[1], ap[1], w, c1);
  ------------------
  |  |   84|  30.1M|  do {                                                                     \
  |  |   85|  30.1M|    register BN_ULONG high, low;                                           \
  |  |   86|  30.1M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  30.1M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  30.1M|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  30.1M|            : "a"(low), "g"(0)                                             \
  |  |   90|  30.1M|            : "cc");                                                       \
  |  |   91|  30.1M|    (r) = (carry);                                                         \
  |  |   92|  30.1M|    (carry) = high;                                                        \
  |  |   93|  30.1M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  141|  30.1M|    mul(rp[2], ap[2], w, c1);
  ------------------
  |  |   84|  30.1M|  do {                                                                     \
  |  |   85|  30.1M|    register BN_ULONG high, low;                                           \
  |  |   86|  30.1M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  30.1M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  30.1M|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  30.1M|            : "a"(low), "g"(0)                                             \
  |  |   90|  30.1M|            : "cc");                                                       \
  |  |   91|  30.1M|    (r) = (carry);                                                         \
  |  |   92|  30.1M|    (carry) = high;                                                        \
  |  |   93|  30.1M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  142|  30.1M|    mul(rp[3], ap[3], w, c1);
  ------------------
  |  |   84|  30.1M|  do {                                                                     \
  |  |   85|  30.1M|    register BN_ULONG high, low;                                           \
  |  |   86|  30.1M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  30.1M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  30.1M|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  30.1M|            : "a"(low), "g"(0)                                             \
  |  |   90|  30.1M|            : "cc");                                                       \
  |  |   91|  30.1M|    (r) = (carry);                                                         \
  |  |   92|  30.1M|    (carry) = high;                                                        \
  |  |   93|  30.1M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  143|  30.1M|    ap += 4;
  144|  30.1M|    rp += 4;
  145|  30.1M|    num -= 4;
  146|  30.1M|  }
  147|  6.94M|  if (num) {
  ------------------
  |  Branch (147:7): [True: 2.79M, False: 4.15M]
  ------------------
  148|  2.79M|    mul(rp[0], ap[0], w, c1);
  ------------------
  |  |   84|  2.79M|  do {                                                                     \
  |  |   85|  2.79M|    register BN_ULONG high, low;                                           \
  |  |   86|  2.79M|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  2.79M|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  2.79M|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  2.79M|            : "a"(low), "g"(0)                                             \
  |  |   90|  2.79M|            : "cc");                                                       \
  |  |   91|  2.79M|    (r) = (carry);                                                         \
  |  |   92|  2.79M|    (carry) = high;                                                        \
  |  |   93|  2.79M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  149|  2.79M|    if (--num == 0) {
  ------------------
  |  Branch (149:9): [True: 2.54M, False: 252k]
  ------------------
  150|  2.54M|      return c1;
  151|  2.54M|    }
  152|   252k|    mul(rp[1], ap[1], w, c1);
  ------------------
  |  |   84|   252k|  do {                                                                     \
  |  |   85|   252k|    register BN_ULONG high, low;                                           \
  |  |   86|   252k|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|   252k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|   252k|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|   252k|            : "a"(low), "g"(0)                                             \
  |  |   90|   252k|            : "cc");                                                       \
  |  |   91|   252k|    (r) = (carry);                                                         \
  |  |   92|   252k|    (carry) = high;                                                        \
  |  |   93|   252k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  153|   252k|    if (--num == 0) {
  ------------------
  |  Branch (153:9): [True: 173k, False: 79.1k]
  ------------------
  154|   173k|      return c1;
  155|   173k|    }
  156|  79.1k|    mul(rp[2], ap[2], w, c1);
  ------------------
  |  |   84|  79.1k|  do {                                                                     \
  |  |   85|  79.1k|    register BN_ULONG high, low;                                           \
  |  |   86|  79.1k|    __asm__("mulq %3" : "=a"(low), "=d"(high) : "a"(word), "g"(a) : "cc"); \
  |  |   87|  79.1k|    __asm__("addq %2,%0; adcq %3,%1"                                       \
  |  |   88|  79.1k|            : "+r"(carry), "+d"(high)                                      \
  |  |   89|  79.1k|            : "a"(low), "g"(0)                                             \
  |  |   90|  79.1k|            : "cc");                                                       \
  |  |   91|  79.1k|    (r) = (carry);                                                         \
  |  |   92|  79.1k|    (carry) = high;                                                        \
  |  |   93|  79.1k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (93:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  157|  79.1k|  }
  158|  4.23M|  return c1;
  159|  6.94M|}
bn_sqr_words:
  161|   358k|void bn_sqr_words(BN_ULONG *r, const BN_ULONG *a, size_t n) {
  162|   358k|  if (n == 0) {
  ------------------
  |  Branch (162:7): [True: 0, False: 358k]
  ------------------
  163|      0|    return;
  164|      0|  }
  165|       |
  166|   382k|  while (n & ~3) {
  ------------------
  |  Branch (166:10): [True: 23.7k, False: 358k]
  ------------------
  167|  23.7k|    sqr(r[0], r[1], a[0]);
  ------------------
  |  |   95|  23.7k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  168|  23.7k|    sqr(r[2], r[3], a[1]);
  ------------------
  |  |   95|  23.7k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  169|  23.7k|    sqr(r[4], r[5], a[2]);
  ------------------
  |  |   95|  23.7k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  170|  23.7k|    sqr(r[6], r[7], a[3]);
  ------------------
  |  |   95|  23.7k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  171|  23.7k|    a += 4;
  172|  23.7k|    r += 8;
  173|  23.7k|    n -= 4;
  174|  23.7k|  }
  175|   358k|  if (n) {
  ------------------
  |  Branch (175:7): [True: 358k, False: 470]
  ------------------
  176|   358k|    sqr(r[0], r[1], a[0]);
  ------------------
  |  |   95|   358k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  177|   358k|    if (--n == 0) {
  ------------------
  |  Branch (177:9): [True: 351k, False: 6.93k]
  ------------------
  178|   351k|      return;
  179|   351k|    }
  180|  6.93k|    sqr(r[2], r[3], a[1]);
  ------------------
  |  |   95|  6.93k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  181|  6.93k|    if (--n == 0) {
  ------------------
  |  Branch (181:9): [True: 3.89k, False: 3.03k]
  ------------------
  182|  3.89k|      return;
  183|  3.89k|    }
  184|  3.03k|    sqr(r[4], r[5], a[2]);
  ------------------
  |  |   95|  3.03k|#define sqr(r0, r1, a) __asm__("mulq %2" : "=a"(r0), "=d"(r1) : "a"(a) : "cc");
  ------------------
  185|  3.03k|  }
  186|   358k|}
bn_add_words:
  189|  3.39M|                      size_t n) {
  190|  3.39M|  BN_ULONG ret;
  191|  3.39M|  size_t i = 0;
  192|       |
  193|  3.39M|  if (n == 0) {
  ------------------
  |  Branch (193:7): [True: 0, False: 3.39M]
  ------------------
  194|      0|    return 0;
  195|      0|  }
  196|       |
  197|  3.39M|  __asm__ volatile (
  198|  3.39M|      "	subq	%0,%0		\n"  // clear carry
  199|  3.39M|      "	jmp	1f		\n"
  200|  3.39M|      ".p2align 4			\n"
  201|  3.39M|      "1:"
  202|  3.39M|      "	movq	(%4,%2,8),%0	\n"
  203|  3.39M|      "	adcq	(%5,%2,8),%0	\n"
  204|  3.39M|      "	movq	%0,(%3,%2,8)	\n"
  205|  3.39M|      "	lea	1(%2),%2	\n"
  206|  3.39M|      "	dec	%1		\n"
  207|  3.39M|      "	jnz	1b		\n"
  208|  3.39M|      "	sbbq	%0,%0		\n"
  209|  3.39M|      : "=&r"(ret), "+c"(n), "+r"(i)
  210|  3.39M|      : "r"(rp), "r"(ap), "r"(bp)
  211|  3.39M|      : "cc", "memory");
  212|       |
  213|  3.39M|  return ret & 1;
  214|  3.39M|}
bn_sub_words:
  217|  11.1M|                      size_t n) {
  218|  11.1M|  BN_ULONG ret;
  219|  11.1M|  size_t i = 0;
  220|       |
  221|  11.1M|  if (n == 0) {
  ------------------
  |  Branch (221:7): [True: 23.2k, False: 11.0M]
  ------------------
  222|  23.2k|    return 0;
  223|  23.2k|  }
  224|       |
  225|  11.0M|  __asm__ volatile (
  226|  11.0M|      "	subq	%0,%0		\n"  // clear borrow
  227|  11.0M|      "	jmp	1f		\n"
  228|  11.0M|      ".p2align 4			\n"
  229|  11.0M|      "1:"
  230|  11.0M|      "	movq	(%4,%2,8),%0	\n"
  231|  11.0M|      "	sbbq	(%5,%2,8),%0	\n"
  232|  11.0M|      "	movq	%0,(%3,%2,8)	\n"
  233|  11.0M|      "	lea	1(%2),%2	\n"
  234|  11.0M|      "	dec	%1		\n"
  235|  11.0M|      "	jnz	1b		\n"
  236|  11.0M|      "	sbbq	%0,%0		\n"
  237|  11.0M|      : "=&r"(ret), "+c"(n), "+r"(i)
  238|  11.0M|      : "r"(rp), "r"(ap), "r"(bp)
  239|  11.0M|      : "cc", "memory");
  240|       |
  241|  11.0M|  return ret & 1;
  242|  11.1M|}
bn_mul_comba8:
  287|  1.71M|void bn_mul_comba8(BN_ULONG r[16], const BN_ULONG a[8], const BN_ULONG b[8]) {
  288|  1.71M|  BN_ULONG c1, c2, c3;
  289|       |
  290|  1.71M|  c1 = 0;
  291|  1.71M|  c2 = 0;
  292|  1.71M|  c3 = 0;
  293|  1.71M|  mul_add_c(a[0], b[0], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  294|  1.71M|  r[0] = c1;
  295|  1.71M|  c1 = 0;
  296|  1.71M|  mul_add_c(a[0], b[1], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  297|  1.71M|  mul_add_c(a[1], b[0], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  298|  1.71M|  r[1] = c2;
  299|  1.71M|  c2 = 0;
  300|  1.71M|  mul_add_c(a[2], b[0], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  301|  1.71M|  mul_add_c(a[1], b[1], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  302|  1.71M|  mul_add_c(a[0], b[2], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  303|  1.71M|  r[2] = c3;
  304|  1.71M|  c3 = 0;
  305|  1.71M|  mul_add_c(a[0], b[3], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  306|  1.71M|  mul_add_c(a[1], b[2], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  307|  1.71M|  mul_add_c(a[2], b[1], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  308|  1.71M|  mul_add_c(a[3], b[0], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  309|  1.71M|  r[3] = c1;
  310|  1.71M|  c1 = 0;
  311|  1.71M|  mul_add_c(a[4], b[0], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  312|  1.71M|  mul_add_c(a[3], b[1], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  313|  1.71M|  mul_add_c(a[2], b[2], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  314|  1.71M|  mul_add_c(a[1], b[3], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  315|  1.71M|  mul_add_c(a[0], b[4], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  316|  1.71M|  r[4] = c2;
  317|  1.71M|  c2 = 0;
  318|  1.71M|  mul_add_c(a[0], b[5], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  319|  1.71M|  mul_add_c(a[1], b[4], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  320|  1.71M|  mul_add_c(a[2], b[3], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  321|  1.71M|  mul_add_c(a[3], b[2], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  322|  1.71M|  mul_add_c(a[4], b[1], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  323|  1.71M|  mul_add_c(a[5], b[0], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  324|  1.71M|  r[5] = c3;
  325|  1.71M|  c3 = 0;
  326|  1.71M|  mul_add_c(a[6], b[0], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  327|  1.71M|  mul_add_c(a[5], b[1], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  328|  1.71M|  mul_add_c(a[4], b[2], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  329|  1.71M|  mul_add_c(a[3], b[3], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  330|  1.71M|  mul_add_c(a[2], b[4], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  331|  1.71M|  mul_add_c(a[1], b[5], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  332|  1.71M|  mul_add_c(a[0], b[6], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  333|  1.71M|  r[6] = c1;
  334|  1.71M|  c1 = 0;
  335|  1.71M|  mul_add_c(a[0], b[7], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  336|  1.71M|  mul_add_c(a[1], b[6], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  337|  1.71M|  mul_add_c(a[2], b[5], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  338|  1.71M|  mul_add_c(a[3], b[4], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  339|  1.71M|  mul_add_c(a[4], b[3], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  340|  1.71M|  mul_add_c(a[5], b[2], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  341|  1.71M|  mul_add_c(a[6], b[1], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  342|  1.71M|  mul_add_c(a[7], b[0], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  343|  1.71M|  r[7] = c2;
  344|  1.71M|  c2 = 0;
  345|  1.71M|  mul_add_c(a[7], b[1], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  346|  1.71M|  mul_add_c(a[6], b[2], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  347|  1.71M|  mul_add_c(a[5], b[3], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  348|  1.71M|  mul_add_c(a[4], b[4], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  349|  1.71M|  mul_add_c(a[3], b[5], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  350|  1.71M|  mul_add_c(a[2], b[6], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  351|  1.71M|  mul_add_c(a[1], b[7], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  352|  1.71M|  r[8] = c3;
  353|  1.71M|  c3 = 0;
  354|  1.71M|  mul_add_c(a[2], b[7], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  355|  1.71M|  mul_add_c(a[3], b[6], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  356|  1.71M|  mul_add_c(a[4], b[5], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  357|  1.71M|  mul_add_c(a[5], b[4], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  358|  1.71M|  mul_add_c(a[6], b[3], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  359|  1.71M|  mul_add_c(a[7], b[2], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  360|  1.71M|  r[9] = c1;
  361|  1.71M|  c1 = 0;
  362|  1.71M|  mul_add_c(a[7], b[3], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  363|  1.71M|  mul_add_c(a[6], b[4], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  364|  1.71M|  mul_add_c(a[5], b[5], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  365|  1.71M|  mul_add_c(a[4], b[6], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  366|  1.71M|  mul_add_c(a[3], b[7], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  367|  1.71M|  r[10] = c2;
  368|  1.71M|  c2 = 0;
  369|  1.71M|  mul_add_c(a[4], b[7], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  370|  1.71M|  mul_add_c(a[5], b[6], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  371|  1.71M|  mul_add_c(a[6], b[5], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  372|  1.71M|  mul_add_c(a[7], b[4], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  373|  1.71M|  r[11] = c3;
  374|  1.71M|  c3 = 0;
  375|  1.71M|  mul_add_c(a[7], b[5], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  376|  1.71M|  mul_add_c(a[6], b[6], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  377|  1.71M|  mul_add_c(a[5], b[7], c1, c2, c3);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  378|  1.71M|  r[12] = c1;
  379|  1.71M|  c1 = 0;
  380|  1.71M|  mul_add_c(a[6], b[7], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  381|  1.71M|  mul_add_c(a[7], b[6], c2, c3, c1);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  382|  1.71M|  r[13] = c2;
  383|  1.71M|  c2 = 0;
  384|  1.71M|  mul_add_c(a[7], b[7], c3, c1, c2);
  ------------------
  |  |  252|  1.71M|  do {                                                               \
  |  |  253|  1.71M|    BN_ULONG t1, t2;                                                 \
  |  |  254|  1.71M|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  255|  1.71M|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  256|  1.71M|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  257|  1.71M|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  258|  1.71M|            : "cc");                                                 \
  |  |  259|  1.71M|  } while (0)
  |  |  ------------------
  |  |  |  Branch (259:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  385|  1.71M|  r[14] = c3;
  386|  1.71M|  r[15] = c1;
  387|  1.71M|}
bn_sqr_comba8:
  427|   172k|void bn_sqr_comba8(BN_ULONG r[16], const BN_ULONG a[8]) {
  428|   172k|  BN_ULONG c1, c2, c3;
  429|       |
  430|   172k|  c1 = 0;
  431|   172k|  c2 = 0;
  432|   172k|  c3 = 0;
  433|   172k|  sqr_add_c(a, 0, c1, c2, c3);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  434|   172k|  r[0] = c1;
  435|   172k|  c1 = 0;
  436|   172k|  sqr_add_c2(a, 1, 0, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  437|   172k|  r[1] = c2;
  438|   172k|  c2 = 0;
  439|   172k|  sqr_add_c(a, 1, c3, c1, c2);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  440|   172k|  sqr_add_c2(a, 2, 0, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  441|   172k|  r[2] = c3;
  442|   172k|  c3 = 0;
  443|   172k|  sqr_add_c2(a, 3, 0, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  444|   172k|  sqr_add_c2(a, 2, 1, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  445|   172k|  r[3] = c1;
  446|   172k|  c1 = 0;
  447|   172k|  sqr_add_c(a, 2, c2, c3, c1);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  448|   172k|  sqr_add_c2(a, 3, 1, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  449|   172k|  sqr_add_c2(a, 4, 0, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  450|   172k|  r[4] = c2;
  451|   172k|  c2 = 0;
  452|   172k|  sqr_add_c2(a, 5, 0, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  453|   172k|  sqr_add_c2(a, 4, 1, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  454|   172k|  sqr_add_c2(a, 3, 2, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  455|   172k|  r[5] = c3;
  456|   172k|  c3 = 0;
  457|   172k|  sqr_add_c(a, 3, c1, c2, c3);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  458|   172k|  sqr_add_c2(a, 4, 2, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  459|   172k|  sqr_add_c2(a, 5, 1, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  460|   172k|  sqr_add_c2(a, 6, 0, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  461|   172k|  r[6] = c1;
  462|   172k|  c1 = 0;
  463|   172k|  sqr_add_c2(a, 7, 0, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  464|   172k|  sqr_add_c2(a, 6, 1, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  465|   172k|  sqr_add_c2(a, 5, 2, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  466|   172k|  sqr_add_c2(a, 4, 3, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  467|   172k|  r[7] = c2;
  468|   172k|  c2 = 0;
  469|   172k|  sqr_add_c(a, 4, c3, c1, c2);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  470|   172k|  sqr_add_c2(a, 5, 3, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  471|   172k|  sqr_add_c2(a, 6, 2, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  472|   172k|  sqr_add_c2(a, 7, 1, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  473|   172k|  r[8] = c3;
  474|   172k|  c3 = 0;
  475|   172k|  sqr_add_c2(a, 7, 2, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  476|   172k|  sqr_add_c2(a, 6, 3, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  477|   172k|  sqr_add_c2(a, 5, 4, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  478|   172k|  r[9] = c1;
  479|   172k|  c1 = 0;
  480|   172k|  sqr_add_c(a, 5, c2, c3, c1);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  481|   172k|  sqr_add_c2(a, 6, 4, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  482|   172k|  sqr_add_c2(a, 7, 3, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  483|   172k|  r[10] = c2;
  484|   172k|  c2 = 0;
  485|   172k|  sqr_add_c2(a, 7, 4, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  486|   172k|  sqr_add_c2(a, 6, 5, c3, c1, c2);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  487|   172k|  r[11] = c3;
  488|   172k|  c3 = 0;
  489|   172k|  sqr_add_c(a, 6, c1, c2, c3);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  490|   172k|  sqr_add_c2(a, 7, 5, c1, c2, c3);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  491|   172k|  r[12] = c1;
  492|   172k|  c1 = 0;
  493|   172k|  sqr_add_c2(a, 7, 6, c2, c3, c1);
  ------------------
  |  |  285|   172k|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|   172k|  do {                                                               \
  |  |  |  |  273|   172k|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|   172k|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|   172k|            : "cc");                                                 \
  |  |  |  |  279|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|   172k|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|   172k|            : "cc");                                                 \
  |  |  |  |  283|   172k|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  494|   172k|  r[13] = c2;
  495|   172k|  c2 = 0;
  496|   172k|  sqr_add_c(a, 7, c3, c1, c2);
  ------------------
  |  |  262|   172k|  do {                                                            \
  |  |  263|   172k|    BN_ULONG t1, t2;                                              \
  |  |  264|   172k|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|   172k|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|   172k|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|   172k|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|   172k|            : "cc");                                              \
  |  |  269|   172k|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  497|   172k|  r[14] = c3;
  498|   172k|  r[15] = c1;
  499|   172k|}
bn_sqr_comba4:
  501|    348|void bn_sqr_comba4(BN_ULONG r[8], const BN_ULONG a[4]) {
  502|    348|  BN_ULONG c1, c2, c3;
  503|       |
  504|    348|  c1 = 0;
  505|    348|  c2 = 0;
  506|    348|  c3 = 0;
  507|    348|  sqr_add_c(a, 0, c1, c2, c3);
  ------------------
  |  |  262|    348|  do {                                                            \
  |  |  263|    348|    BN_ULONG t1, t2;                                              \
  |  |  264|    348|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|    348|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|    348|            : "cc");                                              \
  |  |  269|    348|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  508|    348|  r[0] = c1;
  509|    348|  c1 = 0;
  510|    348|  sqr_add_c2(a, 1, 0, c2, c3, c1);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  511|    348|  r[1] = c2;
  512|    348|  c2 = 0;
  513|    348|  sqr_add_c(a, 1, c3, c1, c2);
  ------------------
  |  |  262|    348|  do {                                                            \
  |  |  263|    348|    BN_ULONG t1, t2;                                              \
  |  |  264|    348|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|    348|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|    348|            : "cc");                                              \
  |  |  269|    348|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  514|    348|  sqr_add_c2(a, 2, 0, c3, c1, c2);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  515|    348|  r[2] = c3;
  516|    348|  c3 = 0;
  517|    348|  sqr_add_c2(a, 3, 0, c1, c2, c3);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  518|    348|  sqr_add_c2(a, 2, 1, c1, c2, c3);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  519|    348|  r[3] = c1;
  520|    348|  c1 = 0;
  521|    348|  sqr_add_c(a, 2, c2, c3, c1);
  ------------------
  |  |  262|    348|  do {                                                            \
  |  |  263|    348|    BN_ULONG t1, t2;                                              \
  |  |  264|    348|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|    348|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|    348|            : "cc");                                              \
  |  |  269|    348|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  522|    348|  sqr_add_c2(a, 3, 1, c2, c3, c1);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  523|    348|  r[4] = c2;
  524|    348|  c2 = 0;
  525|    348|  sqr_add_c2(a, 3, 2, c3, c1, c2);
  ------------------
  |  |  285|    348|#define sqr_add_c2(a, i, j, c0, c1, c2) mul_add_c2((a)[i], (a)[j], c0, c1, c2)
  |  |  ------------------
  |  |  |  |  272|    348|  do {                                                               \
  |  |  |  |  273|    348|    BN_ULONG t1, t2;                                                 \
  |  |  |  |  274|    348|    __asm__("mulq %3" : "=a"(t1), "=d"(t2) : "a"(a), "m"(b) : "cc"); \
  |  |  |  |  275|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  276|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  277|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  278|    348|            : "cc");                                                 \
  |  |  |  |  279|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                     \
  |  |  |  |  280|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                           \
  |  |  |  |  281|    348|            : "r"(t1), "r"(t2), "g"(0)                               \
  |  |  |  |  282|    348|            : "cc");                                                 \
  |  |  |  |  283|    348|  } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (283:12): [Folded - Ignored]
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  526|    348|  r[5] = c3;
  527|    348|  c3 = 0;
  528|    348|  sqr_add_c(a, 3, c1, c2, c3);
  ------------------
  |  |  262|    348|  do {                                                            \
  |  |  263|    348|    BN_ULONG t1, t2;                                              \
  |  |  264|    348|    __asm__("mulq %2" : "=a"(t1), "=d"(t2) : "a"((a)[i]) : "cc"); \
  |  |  265|    348|    __asm__("addq %3,%0; adcq %4,%1; adcq %5,%2"                  \
  |  |  266|    348|            : "+r"(c0), "+r"(c1), "+r"(c2)                        \
  |  |  267|    348|            : "r"(t1), "r"(t2), "g"(0)                            \
  |  |  268|    348|            : "cc");                                              \
  |  |  269|    348|  } while (0)
  |  |  ------------------
  |  |  |  Branch (269:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  529|    348|  r[6] = c1;
  530|    348|  r[7] = c2;
  531|    348|}

BN_new:
   75|  50.1k|BIGNUM *BN_new(void) {
   76|  50.1k|  BIGNUM *bn = OPENSSL_malloc(sizeof(BIGNUM));
   77|       |
   78|  50.1k|  if (bn == NULL) {
  ------------------
  |  Branch (78:7): [True: 0, False: 50.1k]
  ------------------
   79|      0|    return NULL;
   80|      0|  }
   81|       |
   82|  50.1k|  OPENSSL_memset(bn, 0, sizeof(BIGNUM));
   83|  50.1k|  bn->flags = BN_FLG_MALLOCED;
  ------------------
  |  | 1026|  50.1k|#define BN_FLG_MALLOCED 0x01
  ------------------
   84|       |
   85|  50.1k|  return bn;
   86|  50.1k|}
BN_init:
   90|  7.65k|void BN_init(BIGNUM *bn) {
   91|  7.65k|  OPENSSL_memset(bn, 0, sizeof(BIGNUM));
   92|  7.65k|}
BN_free:
   94|  57.8k|void BN_free(BIGNUM *bn) {
   95|  57.8k|  if (bn == NULL) {
  ------------------
  |  Branch (95:7): [True: 0, False: 57.8k]
  ------------------
   96|      0|    return;
   97|      0|  }
   98|       |
   99|  57.8k|  if ((bn->flags & BN_FLG_STATIC_DATA) == 0) {
  ------------------
  |  | 1027|  57.8k|#define BN_FLG_STATIC_DATA 0x02
  ------------------
  |  Branch (99:7): [True: 57.8k, False: 0]
  ------------------
  100|  57.8k|    OPENSSL_free(bn->d);
  101|  57.8k|  }
  102|       |
  103|  57.8k|  if (bn->flags & BN_FLG_MALLOCED) {
  ------------------
  |  | 1026|  57.8k|#define BN_FLG_MALLOCED 0x01
  ------------------
  |  Branch (103:7): [True: 50.1k, False: 7.65k]
  ------------------
  104|  50.1k|    OPENSSL_free(bn);
  105|  50.1k|  } else {
  106|  7.65k|    bn->d = NULL;
  107|  7.65k|  }
  108|  57.8k|}
BN_dup:
  114|  2.74k|BIGNUM *BN_dup(const BIGNUM *src) {
  115|  2.74k|  BIGNUM *copy;
  116|       |
  117|  2.74k|  if (src == NULL) {
  ------------------
  |  Branch (117:7): [True: 0, False: 2.74k]
  ------------------
  118|      0|    return NULL;
  119|      0|  }
  120|       |
  121|  2.74k|  copy = BN_new();
  122|  2.74k|  if (copy == NULL) {
  ------------------
  |  Branch (122:7): [True: 0, False: 2.74k]
  ------------------
  123|      0|    return NULL;
  124|      0|  }
  125|       |
  126|  2.74k|  if (!BN_copy(copy, src)) {
  ------------------
  |  Branch (126:7): [True: 0, False: 2.74k]
  ------------------
  127|      0|    BN_free(copy);
  128|      0|    return NULL;
  129|      0|  }
  130|       |
  131|  2.74k|  return copy;
  132|  2.74k|}
BN_copy:
  134|   619k|BIGNUM *BN_copy(BIGNUM *dest, const BIGNUM *src) {
  135|   619k|  if (src == dest) {
  ------------------
  |  Branch (135:7): [True: 1.06k, False: 618k]
  ------------------
  136|  1.06k|    return dest;
  137|  1.06k|  }
  138|       |
  139|   618k|  if (!bn_wexpand(dest, src->width)) {
  ------------------
  |  Branch (139:7): [True: 0, False: 618k]
  ------------------
  140|      0|    return NULL;
  141|      0|  }
  142|       |
  143|   618k|  OPENSSL_memcpy(dest->d, src->d, sizeof(src->d[0]) * src->width);
  144|       |
  145|   618k|  dest->width = src->width;
  146|   618k|  dest->neg = src->neg;
  147|   618k|  return dest;
  148|   618k|}
BN_num_bits_word:
  170|  1.05M|unsigned BN_num_bits_word(BN_ULONG l) {
  171|       |  // |BN_num_bits| is often called on RSA prime factors. These have public bit
  172|       |  // lengths, but all bits beyond the high bit are secret, so count bits in
  173|       |  // constant time.
  174|  1.05M|  BN_ULONG x, mask;
  175|  1.05M|  int bits = (l != 0);
  176|       |
  177|  1.05M|#if BN_BITS2 > 32
  178|       |  // Look at the upper half of |x|. |x| is at most 64 bits long.
  179|  1.05M|  x = l >> 32;
  180|       |  // Set |mask| to all ones if |x| (the top 32 bits of |l|) is non-zero and all
  181|       |  // all zeros otherwise.
  182|  1.05M|  mask = 0u - x;
  183|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  184|       |  // If |x| is non-zero, the lower half is included in the bit count in full,
  185|       |  // and we count the upper half. Otherwise, we count the lower half.
  186|  1.05M|  bits += 32 & mask;
  187|  1.05M|  l ^= (x ^ l) & mask;  // |l| is |x| if |mask| and remains |l| otherwise.
  188|  1.05M|#endif
  189|       |
  190|       |  // The remaining blocks are analogous iterations at lower powers of two.
  191|  1.05M|  x = l >> 16;
  192|  1.05M|  mask = 0u - x;
  193|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  194|  1.05M|  bits += 16 & mask;
  195|  1.05M|  l ^= (x ^ l) & mask;
  196|       |
  197|  1.05M|  x = l >> 8;
  198|  1.05M|  mask = 0u - x;
  199|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  200|  1.05M|  bits += 8 & mask;
  201|  1.05M|  l ^= (x ^ l) & mask;
  202|       |
  203|  1.05M|  x = l >> 4;
  204|  1.05M|  mask = 0u - x;
  205|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  206|  1.05M|  bits += 4 & mask;
  207|  1.05M|  l ^= (x ^ l) & mask;
  208|       |
  209|  1.05M|  x = l >> 2;
  210|  1.05M|  mask = 0u - x;
  211|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  212|  1.05M|  bits += 2 & mask;
  213|  1.05M|  l ^= (x ^ l) & mask;
  214|       |
  215|  1.05M|  x = l >> 1;
  216|  1.05M|  mask = 0u - x;
  217|  1.05M|  mask = (0u - (mask >> (BN_BITS2 - 1)));
  ------------------
  |  |  151|  1.05M|#define BN_BITS2 64
  ------------------
  218|  1.05M|  bits += 1 & mask;
  219|       |
  220|  1.05M|  return bits;
  221|  1.05M|}
BN_num_bits:
  223|   715k|unsigned BN_num_bits(const BIGNUM *bn) {
  224|   715k|  const int width = bn_minimal_width(bn);
  225|   715k|  if (width == 0) {
  ------------------
  |  Branch (225:7): [True: 1.09k, False: 714k]
  ------------------
  226|  1.09k|    return 0;
  227|  1.09k|  }
  228|       |
  229|   714k|  return (width - 1) * BN_BITS2 + BN_num_bits_word(bn->d[width - 1]);
  ------------------
  |  |  151|   714k|#define BN_BITS2 64
  ------------------
  230|   715k|}
BN_num_bytes:
  232|  5.55k|unsigned BN_num_bytes(const BIGNUM *bn) {
  233|  5.55k|  return (BN_num_bits(bn) + 7) / 8;
  234|  5.55k|}
BN_zero:
  236|  5.23M|void BN_zero(BIGNUM *bn) {
  237|  5.23M|  bn->width = bn->neg = 0;
  238|  5.23M|}
BN_one:
  240|  4.54k|int BN_one(BIGNUM *bn) {
  241|  4.54k|  return BN_set_word(bn, 1);
  242|  4.54k|}
BN_set_word:
  244|  9.75k|int BN_set_word(BIGNUM *bn, BN_ULONG value) {
  245|  9.75k|  if (value == 0) {
  ------------------
  |  Branch (245:7): [True: 0, False: 9.75k]
  ------------------
  246|      0|    BN_zero(bn);
  247|      0|    return 1;
  248|      0|  }
  249|       |
  250|  9.75k|  if (!bn_wexpand(bn, 1)) {
  ------------------
  |  Branch (250:7): [True: 0, False: 9.75k]
  ------------------
  251|      0|    return 0;
  252|      0|  }
  253|       |
  254|  9.75k|  bn->neg = 0;
  255|  9.75k|  bn->d[0] = value;
  256|  9.75k|  bn->width = 1;
  257|  9.75k|  return 1;
  258|  9.75k|}
bn_fits_in_words:
  306|  2.79M|int bn_fits_in_words(const BIGNUM *bn, size_t num) {
  307|       |  // All words beyond |num| must be zero.
  308|  2.79M|  BN_ULONG mask = 0;
  309|  21.9M|  for (size_t i = num; i < (size_t)bn->width; i++) {
  ------------------
  |  Branch (309:24): [True: 19.1M, False: 2.79M]
  ------------------
  310|  19.1M|    mask |= bn->d[i];
  311|  19.1M|  }
  312|  2.79M|  return mask == 0;
  313|  2.79M|}
bn_copy_words:
  315|  9.96k|int bn_copy_words(BN_ULONG *out, size_t num, const BIGNUM *bn) {
  316|  9.96k|  if (bn->neg) {
  ------------------
  |  Branch (316:7): [True: 0, False: 9.96k]
  ------------------
  317|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  318|      0|    return 0;
  319|      0|  }
  320|       |
  321|  9.96k|  size_t width = (size_t)bn->width;
  322|  9.96k|  if (width > num) {
  ------------------
  |  Branch (322:7): [True: 0, False: 9.96k]
  ------------------
  323|      0|    if (!bn_fits_in_words(bn, num)) {
  ------------------
  |  Branch (323:9): [True: 0, False: 0]
  ------------------
  324|      0|      OPENSSL_PUT_ERROR(BN, BN_R_BIGNUM_TOO_LONG);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  325|      0|      return 0;
  326|      0|    }
  327|      0|    width = num;
  328|      0|  }
  329|       |
  330|  9.96k|  OPENSSL_memset(out, 0, sizeof(BN_ULONG) * num);
  331|  9.96k|  OPENSSL_memcpy(out, bn->d, sizeof(BN_ULONG) * width);
  332|  9.96k|  return 1;
  333|  9.96k|}
BN_is_negative:
  335|  5.83k|int BN_is_negative(const BIGNUM *bn) {
  336|  5.83k|  return bn->neg != 0;
  337|  5.83k|}
BN_set_negative:
  339|  2.78k|void BN_set_negative(BIGNUM *bn, int sign) {
  340|  2.78k|  if (sign && !BN_is_zero(bn)) {
  ------------------
  |  Branch (340:7): [True: 2.03k, False: 757]
  |  Branch (340:15): [True: 1.95k, False: 73]
  ------------------
  341|  1.95k|    bn->neg = 1;
  342|  1.95k|  } else {
  343|    830|    bn->neg = 0;
  344|    830|  }
  345|  2.78k|}
bn_wexpand:
  347|  8.93M|int bn_wexpand(BIGNUM *bn, size_t words) {
  348|  8.93M|  BN_ULONG *a;
  349|       |
  350|  8.93M|  if (words <= (size_t)bn->dmax) {
  ------------------
  |  Branch (350:7): [True: 8.84M, False: 81.7k]
  ------------------
  351|  8.84M|    return 1;
  352|  8.84M|  }
  353|       |
  354|  81.7k|  if (words > BN_MAX_WORDS) {
  ------------------
  |  |   73|  81.7k|#define BN_MAX_WORDS (INT_MAX / (4 * BN_BITS2))
  |  |  ------------------
  |  |  |  |  151|  81.7k|#define BN_BITS2 64
  |  |  ------------------
  ------------------
  |  Branch (354:7): [True: 0, False: 81.7k]
  ------------------
  355|      0|    OPENSSL_PUT_ERROR(BN, BN_R_BIGNUM_TOO_LONG);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  356|      0|    return 0;
  357|      0|  }
  358|       |
  359|  81.7k|  if (bn->flags & BN_FLG_STATIC_DATA) {
  ------------------
  |  | 1027|  81.7k|#define BN_FLG_STATIC_DATA 0x02
  ------------------
  |  Branch (359:7): [True: 0, False: 81.7k]
  ------------------
  360|      0|    OPENSSL_PUT_ERROR(BN, BN_R_EXPAND_ON_STATIC_BIGNUM_DATA);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  361|      0|    return 0;
  362|      0|  }
  363|       |
  364|  81.7k|  a = OPENSSL_malloc(sizeof(BN_ULONG) * words);
  365|  81.7k|  if (a == NULL) {
  ------------------
  |  Branch (365:7): [True: 0, False: 81.7k]
  ------------------
  366|      0|    return 0;
  367|      0|  }
  368|       |
  369|  81.7k|  OPENSSL_memcpy(a, bn->d, sizeof(BN_ULONG) * bn->width);
  370|       |
  371|  81.7k|  OPENSSL_free(bn->d);
  372|  81.7k|  bn->d = a;
  373|  81.7k|  bn->dmax = (int)words;
  374|       |
  375|  81.7k|  return 1;
  376|  81.7k|}
bn_resize_words:
  386|   361k|int bn_resize_words(BIGNUM *bn, size_t words) {
  387|   361k|  if ((size_t)bn->width <= words) {
  ------------------
  |  Branch (387:7): [True: 361k, False: 15]
  ------------------
  388|   361k|    if (!bn_wexpand(bn, words)) {
  ------------------
  |  Branch (388:9): [True: 0, False: 361k]
  ------------------
  389|      0|      return 0;
  390|      0|    }
  391|   361k|    OPENSSL_memset(bn->d + bn->width, 0,
  392|   361k|                   (words - bn->width) * sizeof(BN_ULONG));
  393|   361k|    bn->width = (int)words;
  394|   361k|    return 1;
  395|   361k|  }
  396|       |
  397|       |  // All words beyond the new width must be zero.
  398|     15|  if (!bn_fits_in_words(bn, words)) {
  ------------------
  |  Branch (398:7): [True: 0, False: 15]
  ------------------
  399|      0|    OPENSSL_PUT_ERROR(BN, BN_R_BIGNUM_TOO_LONG);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  400|      0|    return 0;
  401|      0|  }
  402|     15|  bn->width = (int)words;
  403|     15|  return 1;
  404|     15|}
bn_select_words:
  407|  2.95M|                     const BN_ULONG *b, size_t num) {
  408|  41.1M|  for (size_t i = 0; i < num; i++) {
  ------------------
  |  Branch (408:22): [True: 38.1M, False: 2.95M]
  ------------------
  409|  38.1M|    static_assert(sizeof(BN_ULONG) <= sizeof(crypto_word_t),
  410|  38.1M|                  "crypto_word_t is too small");
  411|  38.1M|    r[i] = constant_time_select_w(mask, a[i], b[i]);
  412|  38.1M|  }
  413|  2.95M|}
bn_minimal_width:
  415|  7.96M|int bn_minimal_width(const BIGNUM *bn) {
  416|  7.96M|  int ret = bn->width;
  417|  20.4M|  while (ret > 0 && bn->d[ret - 1] == 0) {
  ------------------
  |  Branch (417:10): [True: 19.9M, False: 469k]
  |  Branch (417:21): [True: 12.4M, False: 7.49M]
  ------------------
  418|  12.4M|    ret--;
  419|  12.4M|  }
  420|  7.96M|  return ret;
  421|  7.96M|}
bn_set_minimal_width:
  423|  5.99M|void bn_set_minimal_width(BIGNUM *bn) {
  424|  5.99M|  bn->width = bn_minimal_width(bn);
  425|  5.99M|  if (bn->width == 0) {
  ------------------
  |  Branch (425:7): [True: 417k, False: 5.57M]
  ------------------
  426|   417k|    bn->neg = 0;
  427|   417k|  }
  428|  5.99M|}

bn_big_endian_to_words:
   65|  8.36k|                            size_t in_len) {
   66|  31.4k|  for (size_t i = 0; i < out_len; i++) {
  ------------------
  |  Branch (66:22): [True: 30.6k, False: 846]
  ------------------
   67|  30.6k|    if (in_len < sizeof(BN_ULONG)) {
  ------------------
  |  Branch (67:9): [True: 7.52k, False: 23.1k]
  ------------------
   68|       |      // Load the last partial word.
   69|  7.52k|      BN_ULONG word = 0;
   70|  19.5k|      for (size_t j = 0; j < in_len; j++) {
  ------------------
  |  Branch (70:26): [True: 12.0k, False: 7.52k]
  ------------------
   71|  12.0k|        word = (word << 8) | in[j];
   72|  12.0k|      }
   73|  7.52k|      in_len = 0;
   74|  7.52k|      out[i] = word;
   75|       |      // Fill the remainder with zeros.
   76|  7.52k|      OPENSSL_memset(out + i + 1, 0, (out_len - i - 1) * sizeof(BN_ULONG));
   77|  7.52k|      break;
   78|  7.52k|    }
   79|       |
   80|  23.1k|    in_len -= sizeof(BN_ULONG);
   81|  23.1k|    out[i] = CRYPTO_load_word_be(in + in_len);
   82|  23.1k|  }
   83|       |
   84|       |  // The caller should have sized the output to avoid truncation.
   85|  8.36k|  assert(in_len == 0);
   86|  8.36k|}
BN_bin2bn:
   88|  8.36k|BIGNUM *BN_bin2bn(const uint8_t *in, size_t len, BIGNUM *ret) {
   89|  8.36k|  BIGNUM *bn = NULL;
   90|  8.36k|  if (ret == NULL) {
  ------------------
  |  Branch (90:7): [True: 8.36k, False: 0]
  ------------------
   91|  8.36k|    bn = BN_new();
   92|  8.36k|    if (bn == NULL) {
  ------------------
  |  Branch (92:9): [True: 0, False: 8.36k]
  ------------------
   93|      0|      return NULL;
   94|      0|    }
   95|  8.36k|    ret = bn;
   96|  8.36k|  }
   97|       |
   98|  8.36k|  if (len == 0) {
  ------------------
  |  Branch (98:7): [True: 0, False: 8.36k]
  ------------------
   99|      0|    ret->width = 0;
  100|      0|    return ret;
  101|      0|  }
  102|       |
  103|  8.36k|  size_t num_words = ((len - 1) / BN_BYTES) + 1;
  ------------------
  |  |  152|  8.36k|#define BN_BYTES 8
  ------------------
  104|  8.36k|  if (!bn_wexpand(ret, num_words)) {
  ------------------
  |  Branch (104:7): [True: 0, False: 8.36k]
  ------------------
  105|      0|    BN_free(bn);
  106|      0|    return NULL;
  107|      0|  }
  108|       |
  109|       |  // |bn_wexpand| must check bounds on |num_words| to write it into
  110|       |  // |ret->dmax|.
  111|  8.36k|  assert(num_words <= INT_MAX);
  112|  8.36k|  ret->width = (int)num_words;
  113|  8.36k|  ret->neg = 0;
  114|       |
  115|  8.36k|  bn_big_endian_to_words(ret->d, ret->width, in, len);
  116|  8.36k|  return ret;
  117|  8.36k|}
bn_words_to_big_endian:
  178|  2.77k|                            size_t in_len) {
  179|       |  // The caller should have selected an output length without truncation.
  180|  2.77k|  assert(fits_in_bytes(in, in_len, out_len));
  181|       |
  182|       |  // We only support little-endian platforms, so the internal representation is
  183|       |  // also little-endian as bytes. We can simply copy it in reverse.
  184|  2.77k|  const uint8_t *bytes = (const uint8_t *)in;
  185|  2.77k|  size_t num_bytes = in_len * sizeof(BN_ULONG);
  186|  2.77k|  if (out_len < num_bytes) {
  ------------------
  |  Branch (186:7): [True: 2.09k, False: 688]
  ------------------
  187|  2.09k|    num_bytes = out_len;
  188|  2.09k|  }
  189|       |
  190|   118k|  for (size_t i = 0; i < num_bytes; i++) {
  ------------------
  |  Branch (190:22): [True: 115k, False: 2.77k]
  ------------------
  191|   115k|    out[out_len - i - 1] = bytes[i];
  192|   115k|  }
  193|       |  // Pad out the rest of the buffer with zeroes.
  194|  2.77k|  OPENSSL_memset(out, 0, out_len - num_bytes);
  195|  2.77k|}
BN_bn2bin:
  197|  2.77k|size_t BN_bn2bin(const BIGNUM *in, uint8_t *out) {
  198|  2.77k|  size_t n = BN_num_bytes(in);
  199|  2.77k|  bn_words_to_big_endian(out, n, in->d, in->width);
  200|  2.77k|  return n;
  201|  2.77k|}
bcm.c:fits_in_bytes:
  155|  2.77k|                         size_t num_bytes) {
  156|  2.77k|  const uint8_t *bytes = (const uint8_t *)words;
  157|  2.77k|  size_t tot_bytes = num_words * sizeof(BN_ULONG);
  158|  2.77k|  uint8_t mask = 0;
  159|  21.4k|  for (size_t i = num_bytes; i < tot_bytes; i++) {
  ------------------
  |  Branch (159:30): [True: 18.6k, False: 2.77k]
  ------------------
  160|  18.6k|    mask |= bytes[i];
  161|  18.6k|  }
  162|  2.77k|  return mask == 0;
  163|  2.77k|}

BN_ucmp:
   99|   269k|int BN_ucmp(const BIGNUM *a, const BIGNUM *b) {
  100|   269k|  return bn_cmp_words_consttime(a->d, a->width, b->d, b->width);
  101|   269k|}
BN_cmp:
  103|  5.33k|int BN_cmp(const BIGNUM *a, const BIGNUM *b) {
  104|  5.33k|  if ((a == NULL) || (b == NULL)) {
  ------------------
  |  Branch (104:7): [True: 0, False: 5.33k]
  |  Branch (104:22): [True: 0, False: 5.33k]
  ------------------
  105|      0|    if (a != NULL) {
  ------------------
  |  Branch (105:9): [True: 0, False: 0]
  ------------------
  106|      0|      return -1;
  107|      0|    } else if (b != NULL) {
  ------------------
  |  Branch (107:16): [True: 0, False: 0]
  ------------------
  108|      0|      return 1;
  109|      0|    } else {
  110|      0|      return 0;
  111|      0|    }
  112|      0|  }
  113|       |
  114|       |  // We do not attempt to process the sign bit in constant time. Negative
  115|       |  // |BIGNUM|s should never occur in crypto, only calculators.
  116|  5.33k|  if (a->neg != b->neg) {
  ------------------
  |  Branch (116:7): [True: 0, False: 5.33k]
  ------------------
  117|      0|    if (a->neg) {
  ------------------
  |  Branch (117:9): [True: 0, False: 0]
  ------------------
  118|      0|      return -1;
  119|      0|    }
  120|      0|    return 1;
  121|      0|  }
  122|       |
  123|  5.33k|  int ret = BN_ucmp(a, b);
  124|  5.33k|  return a->neg ? -ret : ret;
  ------------------
  |  Branch (124:10): [True: 0, False: 5.33k]
  ------------------
  125|  5.33k|}
BN_abs_is_word:
  131|  3.15k|int BN_abs_is_word(const BIGNUM *bn, BN_ULONG w) {
  132|  3.15k|  if (bn->width == 0) {
  ------------------
  |  Branch (132:7): [True: 0, False: 3.15k]
  ------------------
  133|      0|    return w == 0;
  134|      0|  }
  135|  3.15k|  BN_ULONG mask = bn->d[0] ^ w;
  136|  19.0k|  for (int i = 1; i < bn->width; i++) {
  ------------------
  |  Branch (136:19): [True: 15.9k, False: 3.15k]
  ------------------
  137|  15.9k|    mask |= bn->d[i];
  138|  15.9k|  }
  139|  3.15k|  return mask == 0;
  140|  3.15k|}
BN_is_zero:
  153|  1.83M|int BN_is_zero(const BIGNUM *bn) {
  154|  1.83M|  return bn_fits_in_words(bn, 0);
  155|  1.83M|}
BN_is_one:
  157|  2.77k|int BN_is_one(const BIGNUM *bn) {
  158|  2.77k|  return bn->neg == 0 && BN_abs_is_word(bn, 1);
  ------------------
  |  Branch (158:10): [True: 2.77k, False: 0]
  |  Branch (158:26): [True: 37, False: 2.74k]
  ------------------
  159|  2.77k|}
BN_is_odd:
  165|   455k|int BN_is_odd(const BIGNUM *bn) {
  166|   455k|  return bn->width > 0 && (bn->d[0] & 1) == 1;
  ------------------
  |  Branch (166:10): [True: 455k, False: 0]
  |  Branch (166:27): [True: 188k, False: 267k]
  ------------------
  167|   455k|}
bcm.c:bn_cmp_words_consttime:
   68|   269k|                                  const BN_ULONG *b, size_t b_len) {
   69|   269k|  static_assert(sizeof(BN_ULONG) <= sizeof(crypto_word_t),
   70|   269k|                "crypto_word_t is too small");
   71|   269k|  int ret = 0;
   72|       |  // Process the common words in little-endian order.
   73|   269k|  size_t min = a_len < b_len ? a_len : b_len;
  ------------------
  |  Branch (73:16): [True: 31.6k, False: 237k]
  ------------------
   74|  2.66M|  for (size_t i = 0; i < min; i++) {
  ------------------
  |  Branch (74:22): [True: 2.39M, False: 269k]
  ------------------
   75|  2.39M|    crypto_word_t eq = constant_time_eq_w(a[i], b[i]);
   76|  2.39M|    crypto_word_t lt = constant_time_lt_w(a[i], b[i]);
   77|  2.39M|    ret =
   78|  2.39M|        constant_time_select_int(eq, ret, constant_time_select_int(lt, -1, 1));
   79|  2.39M|  }
   80|       |
   81|       |  // If |a| or |b| has non-zero words beyond |min|, they take precedence.
   82|   269k|  if (a_len < b_len) {
  ------------------
  |  Branch (82:7): [True: 31.6k, False: 237k]
  ------------------
   83|  31.6k|    crypto_word_t mask = 0;
   84|   117k|    for (size_t i = a_len; i < b_len; i++) {
  ------------------
  |  Branch (84:28): [True: 85.9k, False: 31.6k]
  ------------------
   85|  85.9k|      mask |= b[i];
   86|  85.9k|    }
   87|  31.6k|    ret = constant_time_select_int(constant_time_is_zero_w(mask), ret, -1);
   88|   237k|  } else if (b_len < a_len) {
  ------------------
  |  Branch (88:14): [True: 40.0k, False: 197k]
  ------------------
   89|  40.0k|    crypto_word_t mask = 0;
   90|   837k|    for (size_t i = b_len; i < a_len; i++) {
  ------------------
  |  Branch (90:28): [True: 797k, False: 40.0k]
  ------------------
   91|   797k|      mask |= a[i];
   92|   797k|    }
   93|  40.0k|    ret = constant_time_select_int(constant_time_is_zero_w(mask), ret, 1);
   94|  40.0k|  }
   95|       |
   96|   269k|  return ret;
   97|   269k|}

BN_CTX_new:
  108|  2.77k|BN_CTX *BN_CTX_new(void) {
  109|  2.77k|  BN_CTX *ret = OPENSSL_malloc(sizeof(BN_CTX));
  110|  2.77k|  if (!ret) {
  ------------------
  |  Branch (110:7): [True: 0, False: 2.77k]
  ------------------
  111|      0|    return NULL;
  112|      0|  }
  113|       |
  114|       |  // Initialise the structure
  115|  2.77k|  ret->bignums = NULL;
  116|  2.77k|  BN_STACK_init(&ret->stack);
  117|  2.77k|  ret->used = 0;
  118|  2.77k|  ret->error = 0;
  119|  2.77k|  ret->defer_error = 0;
  120|  2.77k|  return ret;
  121|  2.77k|}
BN_CTX_free:
  123|  4.05k|void BN_CTX_free(BN_CTX *ctx) {
  124|  4.05k|  if (ctx == NULL) {
  ------------------
  |  Branch (124:7): [True: 1.27k, False: 2.77k]
  ------------------
  125|  1.27k|    return;
  126|  1.27k|  }
  127|       |
  128|       |  // All |BN_CTX_start| calls must be matched with |BN_CTX_end|, otherwise the
  129|       |  // function may use more memory than expected, potentially without bound if
  130|       |  // done in a loop. Assert that all |BIGNUM|s have been released.
  131|  2.77k|  assert(ctx->used == 0 || ctx->error);
  132|  2.77k|  sk_BIGNUM_pop_free(ctx->bignums, BN_free);
  133|  2.77k|  BN_STACK_cleanup(&ctx->stack);
  134|  2.77k|  OPENSSL_free(ctx);
  135|  2.77k|}
BN_CTX_start:
  137|  2.87M|void BN_CTX_start(BN_CTX *ctx) {
  138|  2.87M|  if (ctx->error) {
  ------------------
  |  Branch (138:7): [True: 0, False: 2.87M]
  ------------------
  139|       |    // Once an operation has failed, |ctx->stack| no longer matches the number
  140|       |    // of |BN_CTX_end| calls to come. Do nothing.
  141|      0|    return;
  142|      0|  }
  143|       |
  144|  2.87M|  if (!BN_STACK_push(&ctx->stack, ctx->used)) {
  ------------------
  |  Branch (144:7): [True: 0, False: 2.87M]
  ------------------
  145|      0|    ctx->error = 1;
  146|       |    // |BN_CTX_start| cannot fail, so defer the error to |BN_CTX_get|.
  147|      0|    ctx->defer_error = 1;
  148|      0|  }
  149|  2.87M|}
BN_CTX_get:
  151|  5.12M|BIGNUM *BN_CTX_get(BN_CTX *ctx) {
  152|       |  // Once any operation has failed, they all do.
  153|  5.12M|  if (ctx->error) {
  ------------------
  |  Branch (153:7): [True: 0, False: 5.12M]
  ------------------
  154|      0|    if (ctx->defer_error) {
  ------------------
  |  Branch (154:9): [True: 0, False: 0]
  ------------------
  155|      0|      OPENSSL_PUT_ERROR(BN, BN_R_TOO_MANY_TEMPORARY_VARIABLES);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  156|      0|      ctx->defer_error = 0;
  157|      0|    }
  158|      0|    return NULL;
  159|      0|  }
  160|       |
  161|  5.12M|  if (ctx->bignums == NULL) {
  ------------------
  |  Branch (161:7): [True: 2.77k, False: 5.11M]
  ------------------
  162|  2.77k|    ctx->bignums = sk_BIGNUM_new_null();
  163|  2.77k|    if (ctx->bignums == NULL) {
  ------------------
  |  Branch (163:9): [True: 0, False: 2.77k]
  ------------------
  164|      0|      ctx->error = 1;
  165|      0|      return NULL;
  166|      0|    }
  167|  2.77k|  }
  168|       |
  169|  5.12M|  if (ctx->used == sk_BIGNUM_num(ctx->bignums)) {
  ------------------
  |  Branch (169:7): [True: 30.7k, False: 5.09M]
  ------------------
  170|  30.7k|    BIGNUM *bn = BN_new();
  171|  30.7k|    if (bn == NULL || !sk_BIGNUM_push(ctx->bignums, bn)) {
  ------------------
  |  Branch (171:9): [True: 0, False: 30.7k]
  |  Branch (171:23): [True: 0, False: 30.7k]
  ------------------
  172|      0|      OPENSSL_PUT_ERROR(BN, BN_R_TOO_MANY_TEMPORARY_VARIABLES);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  173|      0|      BN_free(bn);
  174|      0|      ctx->error = 1;
  175|      0|      return NULL;
  176|      0|    }
  177|  30.7k|  }
  178|       |
  179|  5.12M|  BIGNUM *ret = sk_BIGNUM_value(ctx->bignums, ctx->used);
  180|  5.12M|  BN_zero(ret);
  181|       |  // This is bounded by |sk_BIGNUM_num|, so it cannot overflow.
  182|  5.12M|  ctx->used++;
  183|  5.12M|  return ret;
  184|  5.12M|}
BN_CTX_end:
  186|  2.87M|void BN_CTX_end(BN_CTX *ctx) {
  187|  2.87M|  if (ctx->error) {
  ------------------
  |  Branch (187:7): [True: 0, False: 2.87M]
  ------------------
  188|       |    // Once an operation has failed, |ctx->stack| no longer matches the number
  189|       |    // of |BN_CTX_end| calls to come. Do nothing.
  190|      0|    return;
  191|      0|  }
  192|       |
  193|  2.87M|  ctx->used = BN_STACK_pop(&ctx->stack);
  194|  2.87M|}
bcm.c:BN_STACK_init:
  199|  2.77k|static void BN_STACK_init(BN_STACK *st) {
  200|  2.77k|  st->indexes = NULL;
  201|  2.77k|  st->depth = st->size = 0;
  202|  2.77k|}
bcm.c:BN_STACK_cleanup:
  204|  2.77k|static void BN_STACK_cleanup(BN_STACK *st) {
  205|  2.77k|  OPENSSL_free(st->indexes);
  206|  2.77k|}
bcm.c:BN_STACK_push:
  208|  2.87M|static int BN_STACK_push(BN_STACK *st, size_t idx) {
  209|  2.87M|  if (st->depth == st->size) {
  ------------------
  |  Branch (209:7): [True: 2.77k, False: 2.86M]
  ------------------
  210|       |    // This function intentionally does not push to the error queue on error.
  211|       |    // Error-reporting is deferred to |BN_CTX_get|.
  212|  2.77k|    size_t new_size = st->size != 0 ? st->size * 3 / 2 : BN_CTX_START_FRAMES;
  ------------------
  |  |   67|  2.77k|#define BN_CTX_START_FRAMES 32
  ------------------
  |  Branch (212:23): [True: 0, False: 2.77k]
  ------------------
  213|  2.77k|    if (new_size <= st->size || new_size > ((size_t)-1) / sizeof(size_t)) {
  ------------------
  |  Branch (213:9): [True: 0, False: 2.77k]
  |  Branch (213:33): [True: 0, False: 2.77k]
  ------------------
  214|      0|      return 0;
  215|      0|    }
  216|  2.77k|    size_t *new_indexes =
  217|  2.77k|        OPENSSL_realloc(st->indexes, new_size * sizeof(size_t));
  218|  2.77k|    if (new_indexes == NULL) {
  ------------------
  |  Branch (218:9): [True: 0, False: 2.77k]
  ------------------
  219|      0|      return 0;
  220|      0|    }
  221|  2.77k|    st->indexes = new_indexes;
  222|  2.77k|    st->size = new_size;
  223|  2.77k|  }
  224|       |
  225|  2.87M|  st->indexes[st->depth] = idx;
  226|  2.87M|  st->depth++;
  227|  2.87M|  return 1;
  228|  2.87M|}
bcm.c:BN_STACK_pop:
  230|  2.87M|static size_t BN_STACK_pop(BN_STACK *st) {
  231|  2.87M|  assert(st->depth > 0);
  232|  2.87M|  st->depth--;
  233|  2.87M|  return st->indexes[st->depth];
  234|  2.87M|}

BN_div:
  195|   624k|           const BIGNUM *divisor, BN_CTX *ctx) {
  196|   624k|  int norm_shift, loop;
  197|   624k|  BIGNUM wnum;
  198|   624k|  BN_ULONG *resp, *wnump;
  199|   624k|  BN_ULONG d0, d1;
  200|   624k|  int num_n, div_n;
  201|       |
  202|       |  // This function relies on the historical minimal-width |BIGNUM| invariant.
  203|       |  // It is already not constant-time (constant-time reductions should use
  204|       |  // Montgomery logic), so we shrink all inputs and intermediate values to
  205|       |  // retain the previous behavior.
  206|       |
  207|       |  // Invalid zero-padding would have particularly bad consequences.
  208|   624k|  int numerator_width = bn_minimal_width(numerator);
  209|   624k|  int divisor_width = bn_minimal_width(divisor);
  210|   624k|  if ((numerator_width > 0 && numerator->d[numerator_width - 1] == 0) ||
  ------------------
  |  Branch (210:8): [True: 573k, False: 51.0k]
  |  Branch (210:31): [True: 0, False: 573k]
  ------------------
  211|   624k|      (divisor_width > 0 && divisor->d[divisor_width - 1] == 0)) {
  ------------------
  |  Branch (211:8): [True: 624k, False: 0]
  |  Branch (211:29): [True: 0, False: 624k]
  ------------------
  212|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NOT_INITIALIZED);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  213|      0|    return 0;
  214|      0|  }
  215|       |
  216|   624k|  if (BN_is_zero(divisor)) {
  ------------------
  |  Branch (216:7): [True: 0, False: 624k]
  ------------------
  217|      0|    OPENSSL_PUT_ERROR(BN, BN_R_DIV_BY_ZERO);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  218|      0|    return 0;
  219|      0|  }
  220|       |
  221|   624k|  BN_CTX_start(ctx);
  222|   624k|  BIGNUM *tmp = BN_CTX_get(ctx);
  223|   624k|  BIGNUM *snum = BN_CTX_get(ctx);
  224|   624k|  BIGNUM *sdiv = BN_CTX_get(ctx);
  225|   624k|  BIGNUM *res = NULL;
  226|   624k|  if (quotient == NULL) {
  ------------------
  |  Branch (226:7): [True: 623k, False: 1.35k]
  ------------------
  227|   623k|    res = BN_CTX_get(ctx);
  228|   623k|  } else {
  229|  1.35k|    res = quotient;
  230|  1.35k|  }
  231|   624k|  if (sdiv == NULL || res == NULL) {
  ------------------
  |  Branch (231:7): [True: 0, False: 624k]
  |  Branch (231:23): [True: 0, False: 624k]
  ------------------
  232|      0|    goto err;
  233|      0|  }
  234|       |
  235|       |  // First we normalise the numbers
  236|   624k|  norm_shift = BN_BITS2 - (BN_num_bits(divisor) % BN_BITS2);
  ------------------
  |  |  151|   624k|#define BN_BITS2 64
  ------------------
                norm_shift = BN_BITS2 - (BN_num_bits(divisor) % BN_BITS2);
  ------------------
  |  |  151|   624k|#define BN_BITS2 64
  ------------------
  237|   624k|  if (!BN_lshift(sdiv, divisor, norm_shift)) {
  ------------------
  |  Branch (237:7): [True: 0, False: 624k]
  ------------------
  238|      0|    goto err;
  239|      0|  }
  240|   624k|  bn_set_minimal_width(sdiv);
  241|   624k|  sdiv->neg = 0;
  242|   624k|  norm_shift += BN_BITS2;
  ------------------
  |  |  151|   624k|#define BN_BITS2 64
  ------------------
  243|   624k|  if (!BN_lshift(snum, numerator, norm_shift)) {
  ------------------
  |  Branch (243:7): [True: 0, False: 624k]
  ------------------
  244|      0|    goto err;
  245|      0|  }
  246|   624k|  bn_set_minimal_width(snum);
  247|   624k|  snum->neg = 0;
  248|       |
  249|       |  // Since we don't want to have special-case logic for the case where snum is
  250|       |  // larger than sdiv, we pad snum with enough zeroes without changing its
  251|       |  // value.
  252|   624k|  if (snum->width <= sdiv->width + 1) {
  ------------------
  |  Branch (252:7): [True: 130k, False: 494k]
  ------------------
  253|   130k|    if (!bn_wexpand(snum, sdiv->width + 2)) {
  ------------------
  |  Branch (253:9): [True: 0, False: 130k]
  ------------------
  254|      0|      goto err;
  255|      0|    }
  256|   516k|    for (int i = snum->width; i < sdiv->width + 2; i++) {
  ------------------
  |  Branch (256:31): [True: 385k, False: 130k]
  ------------------
  257|   385k|      snum->d[i] = 0;
  258|   385k|    }
  259|   130k|    snum->width = sdiv->width + 2;
  260|   494k|  } else {
  261|   494k|    if (!bn_wexpand(snum, snum->width + 1)) {
  ------------------
  |  Branch (261:9): [True: 0, False: 494k]
  ------------------
  262|      0|      goto err;
  263|      0|    }
  264|   494k|    snum->d[snum->width] = 0;
  265|   494k|    snum->width++;
  266|   494k|  }
  267|       |
  268|   624k|  div_n = sdiv->width;
  269|   624k|  num_n = snum->width;
  270|   624k|  loop = num_n - div_n;
  271|       |  // Lets setup a 'window' into snum
  272|       |  // This is the part that corresponds to the current
  273|       |  // 'area' being divided
  274|   624k|  wnum.neg = 0;
  275|   624k|  wnum.d = &(snum->d[loop]);
  276|   624k|  wnum.width = div_n;
  277|       |  // only needed when BN_ucmp messes up the values between width and max
  278|   624k|  wnum.dmax = snum->dmax - loop;  // so we don't step out of bounds
  279|       |
  280|       |  // Get the top 2 words of sdiv
  281|       |  // div_n=sdiv->width;
  282|   624k|  d0 = sdiv->d[div_n - 1];
  283|   624k|  d1 = (div_n == 1) ? 0 : sdiv->d[div_n - 2];
  ------------------
  |  Branch (283:8): [True: 226k, False: 398k]
  ------------------
  284|       |
  285|       |  // pointer to the 'top' of snum
  286|   624k|  wnump = &(snum->d[num_n - 1]);
  287|       |
  288|       |  // Setup |res|. |numerator| and |res| may alias, so we save |numerator->neg|
  289|       |  // for later.
  290|   624k|  const int numerator_neg = numerator->neg;
  291|   624k|  res->neg = (numerator_neg ^ divisor->neg);
  292|   624k|  if (!bn_wexpand(res, loop + 1)) {
  ------------------
  |  Branch (292:7): [True: 0, False: 624k]
  ------------------
  293|      0|    goto err;
  294|      0|  }
  295|   624k|  res->width = loop - 1;
  296|   624k|  resp = &(res->d[loop - 1]);
  297|       |
  298|       |  // space for temp
  299|   624k|  if (!bn_wexpand(tmp, div_n + 1)) {
  ------------------
  |  Branch (299:7): [True: 0, False: 624k]
  ------------------
  300|      0|    goto err;
  301|      0|  }
  302|       |
  303|       |  // if res->width == 0 then clear the neg value otherwise decrease
  304|       |  // the resp pointer
  305|   624k|  if (res->width == 0) {
  ------------------
  |  Branch (305:7): [True: 0, False: 624k]
  ------------------
  306|      0|    res->neg = 0;
  307|   624k|  } else {
  308|   624k|    resp--;
  309|   624k|  }
  310|       |
  311|  7.13M|  for (int i = 0; i < loop - 1; i++, wnump--, resp--) {
  ------------------
  |  Branch (311:19): [True: 6.51M, False: 624k]
  ------------------
  312|  6.51M|    BN_ULONG q, l0;
  313|       |    // the first part of the loop uses the top two words of snum and sdiv to
  314|       |    // calculate a BN_ULONG q such that | wnum - sdiv * q | < sdiv
  315|  6.51M|    BN_ULONG n0, n1, rm = 0;
  316|       |
  317|  6.51M|    n0 = wnump[0];
  318|  6.51M|    n1 = wnump[-1];
  319|  6.51M|    if (n0 == d0) {
  ------------------
  |  Branch (319:9): [True: 4.84k, False: 6.50M]
  ------------------
  320|  4.84k|      q = BN_MASK2;
  ------------------
  |  |  154|  4.84k|#define BN_MASK2 (0xffffffffffffffffUL)
  ------------------
  321|  6.50M|    } else {
  322|       |      // n0 < d0
  323|  6.50M|      bn_div_rem_words(&q, &rm, n0, n1, d0);
  324|       |
  325|  6.50M|#ifdef BN_ULLONG
  326|  6.50M|      BN_ULLONG t2 = (BN_ULLONG)d1 * q;
  ------------------
  |  |  145|  6.50M|#define BN_ULLONG uint128_t
  ------------------
  327|  7.12M|      for (;;) {
  328|  7.12M|        if (t2 <= ((((BN_ULLONG)rm) << BN_BITS2) | wnump[-2])) {
  ------------------
  |  |  151|  7.12M|#define BN_BITS2 64
  ------------------
  |  Branch (328:13): [True: 5.40M, False: 1.72M]
  ------------------
  329|  5.40M|          break;
  330|  5.40M|        }
  331|  1.72M|        q--;
  332|  1.72M|        rm += d0;
  333|  1.72M|        if (rm < d0) {
  ------------------
  |  Branch (333:13): [True: 1.10M, False: 618k]
  ------------------
  334|  1.10M|          break;  // don't let rm overflow
  335|  1.10M|        }
  336|   618k|        t2 -= d1;
  337|   618k|      }
  338|       |#else  // !BN_ULLONG
  339|       |      BN_ULONG t2l, t2h;
  340|       |      BN_UMULT_LOHI(t2l, t2h, d1, q);
  341|       |      for (;;) {
  342|       |        if (t2h < rm ||
  343|       |            (t2h == rm && t2l <= wnump[-2])) {
  344|       |          break;
  345|       |        }
  346|       |        q--;
  347|       |        rm += d0;
  348|       |        if (rm < d0) {
  349|       |          break;  // don't let rm overflow
  350|       |        }
  351|       |        if (t2l < d1) {
  352|       |          t2h--;
  353|       |        }
  354|       |        t2l -= d1;
  355|       |      }
  356|       |#endif  // !BN_ULLONG
  357|  6.50M|    }
  358|       |
  359|  6.51M|    l0 = bn_mul_words(tmp->d, sdiv->d, div_n, q);
  360|  6.51M|    tmp->d[div_n] = l0;
  361|  6.51M|    wnum.d--;
  362|       |    // ingore top values of the bignums just sub the two
  363|       |    // BN_ULONG arrays with bn_sub_words
  364|  6.51M|    if (bn_sub_words(wnum.d, wnum.d, tmp->d, div_n + 1)) {
  ------------------
  |  Branch (364:9): [True: 4.06k, False: 6.50M]
  ------------------
  365|       |      // Note: As we have considered only the leading
  366|       |      // two BN_ULONGs in the calculation of q, sdiv * q
  367|       |      // might be greater than wnum (but then (q-1) * sdiv
  368|       |      // is less or equal than wnum)
  369|  4.06k|      q--;
  370|  4.06k|      if (bn_add_words(wnum.d, wnum.d, sdiv->d, div_n)) {
  ------------------
  |  Branch (370:11): [True: 4.06k, False: 0]
  ------------------
  371|       |        // we can't have an overflow here (assuming
  372|       |        // that q != 0, but if q == 0 then tmp is
  373|       |        // zero anyway)
  374|  4.06k|        (*wnump)++;
  375|  4.06k|      }
  376|  4.06k|    }
  377|       |    // store part of the result
  378|  6.51M|    *resp = q;
  379|  6.51M|  }
  380|       |
  381|   624k|  bn_set_minimal_width(snum);
  382|       |
  383|   624k|  if (rem != NULL) {
  ------------------
  |  Branch (383:7): [True: 623k, False: 1.35k]
  ------------------
  384|   623k|    if (!BN_rshift(rem, snum, norm_shift)) {
  ------------------
  |  Branch (384:9): [True: 0, False: 623k]
  ------------------
  385|      0|      goto err;
  386|      0|    }
  387|   623k|    if (!BN_is_zero(rem)) {
  ------------------
  |  Branch (387:9): [True: 571k, False: 51.3k]
  ------------------
  388|   571k|      rem->neg = numerator_neg;
  389|   571k|    }
  390|   623k|  }
  391|       |
  392|   624k|  bn_set_minimal_width(res);
  393|   624k|  BN_CTX_end(ctx);
  394|   624k|  return 1;
  395|       |
  396|      0|err:
  397|      0|  BN_CTX_end(ctx);
  398|      0|  return 0;
  399|   624k|}
BN_nnmod:
  401|   621k|int BN_nnmod(BIGNUM *r, const BIGNUM *m, const BIGNUM *d, BN_CTX *ctx) {
  402|   621k|  if (!(BN_mod(r, m, d, ctx))) {
  ------------------
  |  |  547|   621k|  BN_div(NULL, (rem), (numerator), (divisor), (ctx))
  ------------------
  |  Branch (402:7): [True: 0, False: 621k]
  ------------------
  403|      0|    return 0;
  404|      0|  }
  405|   621k|  if (!r->neg) {
  ------------------
  |  Branch (405:7): [True: 617k, False: 4.65k]
  ------------------
  406|   617k|    return 1;
  407|   617k|  }
  408|       |
  409|       |  // now -|d| < r < 0, so we have to set r := r + |d|.
  410|  4.65k|  return (d->neg ? BN_sub : BN_add)(r, r, d);
  ------------------
  |  Branch (410:11): [True: 0, False: 4.65k]
  ------------------
  411|   621k|}
bn_reduce_once:
  414|   356k|                        const BN_ULONG *m, size_t num) {
  415|   356k|  assert(r != a);
  416|       |  // |r| = |a| - |m|. |bn_sub_words| performs the bulk of the subtraction, and
  417|       |  // then we apply the borrow to |carry|.
  418|   356k|  carry -= bn_sub_words(r, a, m, num);
  419|       |  // We know 0 <= |a| < 2*|m|, so -|m| <= |r| < |m|.
  420|       |  //
  421|       |  // If 0 <= |r| < |m|, |r| fits in |num| words and |carry| is zero. We then
  422|       |  // wish to select |r| as the answer. Otherwise -m <= r < 0 and we wish to
  423|       |  // return |r| + |m|, or |a|. |carry| must then be -1 or all ones. In both
  424|       |  // cases, |carry| is a suitable input to |bn_select_words|.
  425|       |  //
  426|       |  // Although |carry| may be one if it was one on input and |bn_sub_words|
  427|       |  // returns zero, this would give |r| > |m|, violating our input assumptions.
  428|   356k|  assert(carry == 0 || carry == (BN_ULONG)-1);
  429|   356k|  bn_select_words(r, carry, a /* r < 0 */, r /* r >= 0 */, num);
  430|   356k|  return carry;
  431|   356k|}
bn_reduce_once_in_place:
  434|   475k|                                 BN_ULONG *tmp, size_t num) {
  435|       |  // See |bn_reduce_once| for why this logic works.
  436|   475k|  carry -= bn_sub_words(tmp, r, m, num);
  437|   475k|  assert(carry == 0 || carry == (BN_ULONG)-1);
  438|   475k|  bn_select_words(r, carry, r /* tmp < 0 */, tmp /* tmp >= 0 */, num);
  439|   475k|  return carry;
  440|   475k|}
bn_mod_add_words:
  452|   475k|                      const BN_ULONG *m, BN_ULONG *tmp, size_t num) {
  453|   475k|  BN_ULONG carry = bn_add_words(r, a, b, num);
  454|   475k|  bn_reduce_once_in_place(r, carry, m, tmp, num);
  455|   475k|}
bn_mod_add_consttime:
  598|   475k|                         const BIGNUM *m, BN_CTX *ctx) {
  599|   475k|  BN_CTX_start(ctx);
  600|   475k|  a = bn_resized_from_ctx(a, m->width, ctx);
  601|   475k|  b = bn_resized_from_ctx(b, m->width, ctx);
  602|   475k|  BIGNUM *tmp = bn_scratch_space_from_ctx(m->width, ctx);
  603|   475k|  int ok = a != NULL && b != NULL && tmp != NULL &&
  ------------------
  |  Branch (603:12): [True: 475k, False: 0]
  |  Branch (603:25): [True: 475k, False: 0]
  |  Branch (603:38): [True: 475k, False: 0]
  ------------------
  604|   475k|           bn_wexpand(r, m->width);
  ------------------
  |  Branch (604:12): [True: 475k, False: 0]
  ------------------
  605|   475k|  if (ok) {
  ------------------
  |  Branch (605:7): [True: 475k, False: 0]
  ------------------
  606|   475k|    bn_mod_add_words(r->d, a->d, b->d, m->d, tmp->d, m->width);
  607|   475k|    r->width = m->width;
  608|   475k|    r->neg = 0;
  609|   475k|  }
  610|   475k|  BN_CTX_end(ctx);
  611|   475k|  return ok;
  612|   475k|}
bn_mod_lshift_consttime:
  713|  1.06k|                            BN_CTX *ctx) {
  714|  1.06k|  if (!BN_copy(r, a)) {
  ------------------
  |  Branch (714:7): [True: 0, False: 1.06k]
  ------------------
  715|      0|    return 0;
  716|      0|  }
  717|   476k|  for (int i = 0; i < n; i++) {
  ------------------
  |  Branch (717:19): [True: 475k, False: 1.06k]
  ------------------
  718|   475k|    if (!bn_mod_lshift1_consttime(r, r, m, ctx)) {
  ------------------
  |  Branch (718:9): [True: 0, False: 475k]
  ------------------
  719|      0|      return 0;
  720|      0|    }
  721|   475k|  }
  722|  1.06k|  return 1;
  723|  1.06k|}
bn_mod_lshift1_consttime:
  742|   475k|                             BN_CTX *ctx) {
  743|   475k|  return bn_mod_add_consttime(r, a, a, m, ctx);
  744|   475k|}
bcm.c:bn_div_rem_words:
  140|  6.50M|                                    BN_ULONG n0, BN_ULONG n1, BN_ULONG d0) {
  141|       |  // GCC and Clang generate function calls to |__udivdi3| and |__umoddi3| when
  142|       |  // the |BN_ULLONG|-based C code is used.
  143|       |  //
  144|       |  // GCC bugs:
  145|       |  //   * https://gcc.gnu.org/bugzilla/show_bug.cgi?id=14224
  146|       |  //   * https://gcc.gnu.org/bugzilla/show_bug.cgi?id=43721
  147|       |  //   * https://gcc.gnu.org/bugzilla/show_bug.cgi?id=54183
  148|       |  //   * https://gcc.gnu.org/bugzilla/show_bug.cgi?id=58897
  149|       |  //   * https://gcc.gnu.org/bugzilla/show_bug.cgi?id=65668
  150|       |  //
  151|       |  // Clang bugs:
  152|       |  //   * https://llvm.org/bugs/show_bug.cgi?id=6397
  153|       |  //   * https://llvm.org/bugs/show_bug.cgi?id=12418
  154|       |  //
  155|       |  // These issues aren't specific to x86 and x86_64, so it might be worthwhile
  156|       |  // to add more assembly language implementations.
  157|       |#if defined(BN_CAN_USE_INLINE_ASM) && defined(OPENSSL_X86)
  158|       |  __asm__ volatile("divl %4"
  159|       |                   : "=a"(*quotient_out), "=d"(*rem_out)
  160|       |                   : "a"(n1), "d"(n0), "rm"(d0)
  161|       |                   : "cc");
  162|       |#elif defined(BN_CAN_USE_INLINE_ASM) && defined(OPENSSL_X86_64)
  163|  6.50M|  __asm__ volatile("divq %4"
  164|  6.50M|                   : "=a"(*quotient_out), "=d"(*rem_out)
  165|  6.50M|                   : "a"(n1), "d"(n0), "rm"(d0)
  166|  6.50M|                   : "cc");
  167|       |#else
  168|       |#if defined(BN_CAN_DIVIDE_ULLONG)
  169|       |  BN_ULLONG n = (((BN_ULLONG)n0) << BN_BITS2) | n1;
  170|       |  *quotient_out = (BN_ULONG)(n / d0);
  171|       |#else
  172|       |  *quotient_out = bn_div_words(n0, n1, d0);
  173|       |#endif
  174|       |  *rem_out = n1 - (*quotient_out * d0);
  175|       |#endif
  176|  6.50M|}
bcm.c:bn_resized_from_ctx:
  565|   951k|                                         BN_CTX *ctx) {
  566|   951k|  if ((size_t)bn->width >= width) {
  ------------------
  |  Branch (566:7): [True: 951k, False: 0]
  ------------------
  567|       |    // Any excess words must be zero.
  568|   951k|    assert(bn_fits_in_words(bn, width));
  569|   951k|    return bn;
  570|   951k|  }
  571|      0|  BIGNUM *ret = bn_scratch_space_from_ctx(width, ctx);
  572|      0|  if (ret == NULL ||
  ------------------
  |  Branch (572:7): [True: 0, False: 0]
  ------------------
  573|      0|      !BN_copy(ret, bn) ||
  ------------------
  |  Branch (573:7): [True: 0, False: 0]
  ------------------
  574|      0|      !bn_resize_words(ret, width)) {
  ------------------
  |  Branch (574:7): [True: 0, False: 0]
  ------------------
  575|      0|    return NULL;
  576|      0|  }
  577|      0|  return ret;
  578|      0|}
bcm.c:bn_scratch_space_from_ctx:
  548|   475k|static BIGNUM *bn_scratch_space_from_ctx(size_t width, BN_CTX *ctx) {
  549|   475k|  BIGNUM *ret = BN_CTX_get(ctx);
  550|   475k|  if (ret == NULL ||
  ------------------
  |  Branch (550:7): [True: 0, False: 475k]
  ------------------
  551|   475k|      !bn_wexpand(ret, width)) {
  ------------------
  |  Branch (551:7): [True: 0, False: 475k]
  ------------------
  552|      0|    return NULL;
  553|      0|  }
  554|   475k|  ret->neg = 0;
  555|   475k|  ret->width = (int)width;
  556|   475k|  return ret;
  557|   475k|}

BN_mod_exp:
  568|  2.77k|               BN_CTX *ctx) {
  569|  2.77k|  if (m->neg) {
  ------------------
  |  Branch (569:7): [True: 0, False: 2.77k]
  ------------------
  570|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  571|      0|    return 0;
  572|      0|  }
  573|  2.77k|  if (a->neg || BN_ucmp(a, m) >= 0) {
  ------------------
  |  Branch (573:7): [True: 1.95k, False: 829]
  |  Branch (573:17): [True: 253, False: 576]
  ------------------
  574|  2.20k|    if (!BN_nnmod(r, a, m, ctx)) {
  ------------------
  |  Branch (574:9): [True: 0, False: 2.20k]
  ------------------
  575|      0|      return 0;
  576|      0|    }
  577|  2.20k|    a = r;
  578|  2.20k|  }
  579|       |
  580|  2.77k|  if (BN_is_odd(m)) {
  ------------------
  |  Branch (580:7): [True: 1.27k, False: 1.50k]
  ------------------
  581|  1.27k|    return BN_mod_exp_mont(r, a, p, m, ctx, NULL);
  582|  1.27k|  }
  583|       |
  584|  1.50k|  return mod_exp_recp(r, a, p, m, ctx);
  585|  2.77k|}
BN_mod_exp_mont:
  588|  2.55k|                    const BIGNUM *m, BN_CTX *ctx, const BN_MONT_CTX *mont) {
  589|  2.55k|  if (!BN_is_odd(m)) {
  ------------------
  |  Branch (589:7): [True: 0, False: 2.55k]
  ------------------
  590|      0|    OPENSSL_PUT_ERROR(BN, BN_R_CALLED_WITH_EVEN_MODULUS);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  591|      0|    return 0;
  592|      0|  }
  593|  2.55k|  if (m->neg) {
  ------------------
  |  Branch (593:7): [True: 0, False: 2.55k]
  ------------------
  594|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  595|      0|    return 0;
  596|      0|  }
  597|       |  // |a| is secret, but |a < m| is not.
  598|  2.55k|  if (a->neg || constant_time_declassify_int(BN_ucmp(a, m)) >= 0) {
  ------------------
  |  Branch (598:7): [True: 0, False: 2.55k]
  |  Branch (598:17): [True: 0, False: 2.55k]
  ------------------
  599|      0|    OPENSSL_PUT_ERROR(BN, BN_R_INPUT_NOT_REDUCED);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  600|      0|    return 0;
  601|      0|  }
  602|       |
  603|  2.55k|  int bits = BN_num_bits(p);
  604|  2.55k|  if (bits == 0) {
  ------------------
  |  Branch (604:7): [True: 372, False: 2.18k]
  ------------------
  605|       |    // x**0 mod 1 is still zero.
  606|    372|    if (BN_abs_is_word(m, 1)) {
  ------------------
  |  Branch (606:9): [True: 14, False: 358]
  ------------------
  607|     14|      BN_zero(rr);
  608|     14|      return 1;
  609|     14|    }
  610|    358|    return BN_one(rr);
  611|    372|  }
  612|       |
  613|  2.18k|  int ret = 0;
  614|  2.18k|  BIGNUM *val[TABLE_SIZE];
  615|  2.18k|  BN_MONT_CTX *new_mont = NULL;
  616|       |
  617|  2.18k|  BN_CTX_start(ctx);
  618|  2.18k|  BIGNUM *r = BN_CTX_get(ctx);
  619|  2.18k|  val[0] = BN_CTX_get(ctx);
  620|  2.18k|  if (r == NULL || val[0] == NULL) {
  ------------------
  |  Branch (620:7): [True: 0, False: 2.18k]
  |  Branch (620:20): [True: 0, False: 2.18k]
  ------------------
  621|      0|    goto err;
  622|      0|  }
  623|       |
  624|       |  // Allocate a montgomery context if it was not supplied by the caller.
  625|  2.18k|  if (mont == NULL) {
  ------------------
  |  Branch (625:7): [True: 1.09k, False: 1.09k]
  ------------------
  626|  1.09k|    new_mont = BN_MONT_CTX_new_consttime(m, ctx);
  627|  1.09k|    if (new_mont == NULL) {
  ------------------
  |  Branch (627:9): [True: 0, False: 1.09k]
  ------------------
  628|      0|      goto err;
  629|      0|    }
  630|  1.09k|    mont = new_mont;
  631|  1.09k|  }
  632|       |
  633|       |  // We exponentiate by looking at sliding windows of the exponent and
  634|       |  // precomputing powers of |a|. Windows may be shifted so they always end on a
  635|       |  // set bit, so only precompute odd powers. We compute val[i] = a^(2*i + 1)
  636|       |  // for i = 0 to 2^(window-1), all in Montgomery form.
  637|  2.18k|  int window = BN_window_bits_for_exponent_size(bits);
  638|  2.18k|  if (!BN_to_montgomery(val[0], a, mont, ctx)) {
  ------------------
  |  Branch (638:7): [True: 0, False: 2.18k]
  ------------------
  639|      0|    goto err;
  640|      0|  }
  641|  2.18k|  if (window > 1) {
  ------------------
  |  Branch (641:7): [True: 1.12k, False: 1.05k]
  ------------------
  642|  1.12k|    BIGNUM *d = BN_CTX_get(ctx);
  643|  1.12k|    if (d == NULL ||
  ------------------
  |  Branch (643:9): [True: 0, False: 1.12k]
  ------------------
  644|  1.12k|        !BN_mod_mul_montgomery(d, val[0], val[0], mont, ctx)) {
  ------------------
  |  Branch (644:9): [True: 0, False: 1.12k]
  ------------------
  645|      0|      goto err;
  646|      0|    }
  647|  19.9k|    for (int i = 1; i < 1 << (window - 1); i++) {
  ------------------
  |  Branch (647:21): [True: 18.8k, False: 1.12k]
  ------------------
  648|  18.8k|      val[i] = BN_CTX_get(ctx);
  649|  18.8k|      if (val[i] == NULL ||
  ------------------
  |  Branch (649:11): [True: 0, False: 18.8k]
  ------------------
  650|  18.8k|          !BN_mod_mul_montgomery(val[i], val[i - 1], d, mont, ctx)) {
  ------------------
  |  Branch (650:11): [True: 0, False: 18.8k]
  ------------------
  651|      0|        goto err;
  652|      0|      }
  653|  18.8k|    }
  654|  1.12k|  }
  655|       |
  656|       |  // |p| is non-zero, so at least one window is non-zero. To save some
  657|       |  // multiplications, defer initializing |r| until then.
  658|  2.18k|  int r_is_one = 1;
  659|  2.18k|  int wstart = bits - 1;  // The top bit of the window.
  660|   435k|  for (;;) {
  661|   435k|    if (!BN_is_bit_set(p, wstart)) {
  ------------------
  |  Branch (661:9): [True: 366k, False: 69.8k]
  ------------------
  662|   366k|      if (!r_is_one && !BN_mod_mul_montgomery(r, r, r, mont, ctx)) {
  ------------------
  |  Branch (662:11): [True: 366k, False: 0]
  |  Branch (662:24): [True: 0, False: 366k]
  ------------------
  663|      0|        goto err;
  664|      0|      }
  665|   366k|      if (wstart == 0) {
  ------------------
  |  Branch (665:11): [True: 936, False: 365k]
  ------------------
  666|    936|        break;
  667|    936|      }
  668|   365k|      wstart--;
  669|   365k|      continue;
  670|   366k|    }
  671|       |
  672|       |    // We now have wstart on a set bit. Find the largest window we can use.
  673|  69.8k|    int wvalue = 1;
  674|  69.8k|    int wsize = 0;
  675|   367k|    for (int i = 1; i < window && i <= wstart; i++) {
  ------------------
  |  Branch (675:21): [True: 298k, False: 69.2k]
  |  Branch (675:35): [True: 297k, False: 576]
  ------------------
  676|   297k|      if (BN_is_bit_set(p, wstart - i)) {
  ------------------
  |  Branch (676:11): [True: 184k, False: 113k]
  ------------------
  677|   184k|        wvalue <<= (i - wsize);
  678|   184k|        wvalue |= 1;
  679|   184k|        wsize = i;
  680|   184k|      }
  681|   297k|    }
  682|       |
  683|       |    // Shift |r| to the end of the window.
  684|  69.8k|    if (!r_is_one) {
  ------------------
  |  Branch (684:9): [True: 67.6k, False: 2.18k]
  ------------------
  685|   357k|      for (int i = 0; i < wsize + 1; i++) {
  ------------------
  |  Branch (685:23): [True: 289k, False: 67.6k]
  ------------------
  686|   289k|        if (!BN_mod_mul_montgomery(r, r, r, mont, ctx)) {
  ------------------
  |  Branch (686:13): [True: 0, False: 289k]
  ------------------
  687|      0|          goto err;
  688|      0|        }
  689|   289k|      }
  690|  67.6k|    }
  691|       |
  692|  69.8k|    assert(wvalue & 1);
  693|  69.8k|    assert(wvalue < (1 << window));
  694|  69.8k|    if (r_is_one) {
  ------------------
  |  Branch (694:9): [True: 2.18k, False: 67.6k]
  ------------------
  695|  2.18k|      if (!BN_copy(r, val[wvalue >> 1])) {
  ------------------
  |  Branch (695:11): [True: 0, False: 2.18k]
  ------------------
  696|      0|        goto err;
  697|      0|      }
  698|  67.6k|    } else if (!BN_mod_mul_montgomery(r, r, val[wvalue >> 1], mont, ctx)) {
  ------------------
  |  Branch (698:16): [True: 0, False: 67.6k]
  ------------------
  699|      0|      goto err;
  700|      0|    }
  701|       |
  702|  69.8k|    r_is_one = 0;
  703|  69.8k|    if (wstart == wsize) {
  ------------------
  |  Branch (703:9): [True: 1.24k, False: 68.5k]
  ------------------
  704|  1.24k|      break;
  705|  1.24k|    }
  706|  68.5k|    wstart -= wsize + 1;
  707|  68.5k|  }
  708|       |
  709|       |  // |p| is non-zero, so |r_is_one| must be cleared at some point.
  710|  2.18k|  assert(!r_is_one);
  711|       |
  712|  2.18k|  if (!BN_from_montgomery(rr, r, mont, ctx)) {
  ------------------
  |  Branch (712:7): [True: 0, False: 2.18k]
  ------------------
  713|      0|    goto err;
  714|      0|  }
  715|  2.18k|  ret = 1;
  716|       |
  717|  2.18k|err:
  718|  2.18k|  BN_MONT_CTX_free(new_mont);
  719|  2.18k|  BN_CTX_end(ctx);
  720|  2.18k|  return ret;
  721|  2.18k|}
BN_mod_exp_mont_consttime:
  885|  1.27k|                              const BN_MONT_CTX *mont) {
  886|  1.27k|  int i, ret = 0, wvalue;
  887|  1.27k|  BN_MONT_CTX *new_mont = NULL;
  888|       |
  889|  1.27k|  unsigned char *powerbuf_free = NULL;
  890|  1.27k|  size_t powerbuf_len = 0;
  891|  1.27k|  BN_ULONG *powerbuf = NULL;
  892|       |
  893|  1.27k|  if (!BN_is_odd(m)) {
  ------------------
  |  Branch (893:7): [True: 0, False: 1.27k]
  ------------------
  894|      0|    OPENSSL_PUT_ERROR(BN, BN_R_CALLED_WITH_EVEN_MODULUS);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  895|      0|    return 0;
  896|      0|  }
  897|  1.27k|  if (m->neg) {
  ------------------
  |  Branch (897:7): [True: 0, False: 1.27k]
  ------------------
  898|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  899|      0|    return 0;
  900|      0|  }
  901|  1.27k|  if (a->neg || BN_ucmp(a, m) >= 0) {
  ------------------
  |  Branch (901:7): [True: 0, False: 1.27k]
  |  Branch (901:17): [True: 0, False: 1.27k]
  ------------------
  902|      0|    OPENSSL_PUT_ERROR(BN, BN_R_INPUT_NOT_REDUCED);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  903|      0|    return 0;
  904|      0|  }
  905|       |
  906|       |  // Use all bits stored in |p|, rather than |BN_num_bits|, so we do not leak
  907|       |  // whether the top bits are zero.
  908|  1.27k|  int max_bits = p->width * BN_BITS2;
  ------------------
  |  |  151|  1.27k|#define BN_BITS2 64
  ------------------
  909|  1.27k|  int bits = max_bits;
  910|  1.27k|  if (bits == 0) {
  ------------------
  |  Branch (910:7): [True: 0, False: 1.27k]
  ------------------
  911|       |    // x**0 mod 1 is still zero.
  912|      0|    if (BN_abs_is_word(m, 1)) {
  ------------------
  |  Branch (912:9): [True: 0, False: 0]
  ------------------
  913|      0|      BN_zero(rr);
  914|      0|      return 1;
  915|      0|    }
  916|      0|    return BN_one(rr);
  917|      0|  }
  918|       |
  919|       |  // Allocate a montgomery context if it was not supplied by the caller.
  920|  1.27k|  if (mont == NULL) {
  ------------------
  |  Branch (920:7): [True: 0, False: 1.27k]
  ------------------
  921|      0|    new_mont = BN_MONT_CTX_new_consttime(m, ctx);
  922|      0|    if (new_mont == NULL) {
  ------------------
  |  Branch (922:9): [True: 0, False: 0]
  ------------------
  923|      0|      goto err;
  924|      0|    }
  925|      0|    mont = new_mont;
  926|      0|  }
  927|       |
  928|       |  // Use the width in |mont->N|, rather than the copy in |m|. The assembly
  929|       |  // implementation assumes it can use |top| to size R.
  930|  1.27k|  int top = mont->N.width;
  931|       |
  932|  1.27k|#if defined(OPENSSL_BN_ASM_MONT5) || defined(RSAZ_ENABLED)
  933|       |  // Share one large stack-allocated buffer between the RSAZ and non-RSAZ code
  934|       |  // paths. If we were to use separate static buffers for each then there is
  935|       |  // some chance that both large buffers would be allocated on the stack,
  936|       |  // causing the stack space requirement to be truly huge (~10KB).
  937|  1.27k|  alignas(MOD_EXP_CTIME_ALIGN) BN_ULONG storage[MOD_EXP_CTIME_STORAGE_LEN];
  938|  1.27k|#endif
  939|  1.27k|#if defined(RSAZ_ENABLED)
  940|       |  // If the size of the operands allow it, perform the optimized RSAZ
  941|       |  // exponentiation. For further information see crypto/fipsmodule/bn/rsaz_exp.c
  942|       |  // and accompanying assembly modules.
  943|  1.27k|  if (a->width == 16 && p->width == 16 && BN_num_bits(m) == 1024 &&
  ------------------
  |  Branch (943:7): [True: 210, False: 1.06k]
  |  Branch (943:25): [True: 202, False: 8]
  |  Branch (943:43): [True: 69, False: 133]
  ------------------
  944|  1.27k|      rsaz_avx2_preferred()) {
  ------------------
  |  Branch (944:7): [True: 0, False: 69]
  ------------------
  945|      0|    if (!bn_wexpand(rr, 16)) {
  ------------------
  |  Branch (945:9): [True: 0, False: 0]
  ------------------
  946|      0|      goto err;
  947|      0|    }
  948|      0|    RSAZ_1024_mod_exp_avx2(rr->d, a->d, p->d, m->d, mont->RR.d, mont->n0[0],
  949|      0|                           storage);
  950|      0|    rr->width = 16;
  951|      0|    rr->neg = 0;
  952|      0|    ret = 1;
  953|      0|    goto err;
  954|      0|  }
  955|  1.27k|#endif
  956|       |
  957|       |  // Get the window size to use with size of p.
  958|  1.27k|  int window = BN_window_bits_for_ctime_exponent_size(bits);
  ------------------
  |  |  876|  1.27k|  ((b) > 937 ? 6 : (b) > 306 ? 5 : (b) > 89 ? 4 : (b) > 22 ? 3 : 1)
  |  |  ------------------
  |  |  |  Branch (876:4): [True: 247, False: 1.03k]
  |  |  |  Branch (876:20): [True: 57, False: 974]
  |  |  |  Branch (876:36): [True: 75, False: 899]
  |  |  |  Branch (876:51): [True: 899, False: 0]
  |  |  ------------------
  ------------------
  959|  1.27k|  assert(window <= BN_MAX_MOD_EXP_CTIME_WINDOW);
  960|       |
  961|       |  // Calculating |powerbuf_len| below cannot overflow because of the bound on
  962|       |  // Montgomery reduction.
  963|  1.27k|  assert((size_t)top <= BN_MONTGOMERY_MAX_WORDS);
  964|  1.27k|  static_assert(
  965|  1.27k|      BN_MONTGOMERY_MAX_WORDS <=
  966|  1.27k|          INT_MAX / sizeof(BN_ULONG) / ((1 << BN_MAX_MOD_EXP_CTIME_WINDOW) + 3),
  967|  1.27k|      "powerbuf_len may overflow");
  968|       |
  969|  1.27k|#if defined(OPENSSL_BN_ASM_MONT5)
  970|  1.27k|  if (window >= 5) {
  ------------------
  |  Branch (970:7): [True: 304, False: 974]
  ------------------
  971|    304|    window = 5;  // ~5% improvement for RSA2048 sign, and even for RSA4096
  972|       |    // Reserve space for the |mont->N| copy.
  973|    304|    powerbuf_len += top * sizeof(mont->N.d[0]);
  974|    304|  }
  975|  1.27k|#endif
  976|       |
  977|       |  // Allocate a buffer large enough to hold all of the pre-computed
  978|       |  // powers of |am|, |am| itself, and |tmp|.
  979|  1.27k|  int num_powers = 1 << window;
  980|  1.27k|  powerbuf_len += sizeof(m->d[0]) * top * (num_powers + 2);
  981|       |
  982|  1.27k|#if defined(OPENSSL_BN_ASM_MONT5)
  983|  1.27k|  if (powerbuf_len <= sizeof(storage)) {
  ------------------
  |  Branch (983:7): [True: 1.27k, False: 8]
  ------------------
  984|  1.27k|    powerbuf = storage;
  985|  1.27k|  }
  986|       |  // |storage| is more than large enough to handle 1024-bit inputs.
  987|  1.27k|  assert(powerbuf != NULL || top * BN_BITS2 > 1024);
  988|  1.27k|#endif
  989|  1.27k|  if (powerbuf == NULL) {
  ------------------
  |  Branch (989:7): [True: 8, False: 1.27k]
  ------------------
  990|      8|    powerbuf_free = OPENSSL_malloc(powerbuf_len + MOD_EXP_CTIME_ALIGN);
  ------------------
  |  |  200|      8|#define MOD_EXP_CTIME_ALIGN 64
  ------------------
  991|      8|    if (powerbuf_free == NULL) {
  ------------------
  |  Branch (991:9): [True: 0, False: 8]
  ------------------
  992|      0|      goto err;
  993|      0|    }
  994|      8|    powerbuf = align_pointer(powerbuf_free, MOD_EXP_CTIME_ALIGN);
  ------------------
  |  |  200|      8|#define MOD_EXP_CTIME_ALIGN 64
  ------------------
  995|      8|  }
  996|  1.27k|  OPENSSL_memset(powerbuf, 0, powerbuf_len);
  997|       |
  998|       |  // Place |tmp| and |am| right after powers table.
  999|  1.27k|  BIGNUM tmp, am;
 1000|  1.27k|  tmp.d = powerbuf + top * num_powers;
 1001|  1.27k|  am.d = tmp.d + top;
 1002|  1.27k|  tmp.width = am.width = 0;
 1003|  1.27k|  tmp.dmax = am.dmax = top;
 1004|  1.27k|  tmp.neg = am.neg = 0;
 1005|  1.27k|  tmp.flags = am.flags = BN_FLG_STATIC_DATA;
  ------------------
  |  | 1027|  1.27k|#define BN_FLG_STATIC_DATA 0x02
  ------------------
 1006|       |
 1007|  1.27k|  if (!bn_one_to_montgomery(&tmp, mont, ctx) ||
  ------------------
  |  Branch (1007:7): [True: 0, False: 1.27k]
  ------------------
 1008|  1.27k|      !bn_resize_words(&tmp, top)) {
  ------------------
  |  Branch (1008:7): [True: 0, False: 1.27k]
  ------------------
 1009|      0|    goto err;
 1010|      0|  }
 1011|       |
 1012|       |  // Prepare a^1 in the Montgomery domain.
 1013|  1.27k|  assert(!a->neg);
 1014|  1.27k|  assert(BN_ucmp(a, m) < 0);
 1015|  1.27k|  if (!BN_to_montgomery(&am, a, mont, ctx) ||
  ------------------
  |  Branch (1015:7): [True: 0, False: 1.27k]
  ------------------
 1016|  1.27k|      !bn_resize_words(&am, top)) {
  ------------------
  |  Branch (1016:7): [True: 0, False: 1.27k]
  ------------------
 1017|      0|    goto err;
 1018|      0|  }
 1019|       |
 1020|  1.27k|#if defined(OPENSSL_BN_ASM_MONT5)
 1021|       |  // This optimization uses ideas from https://eprint.iacr.org/2011/239,
 1022|       |  // specifically optimization of cache-timing attack countermeasures,
 1023|       |  // pre-computation optimization, and Almost Montgomery Multiplication.
 1024|       |  //
 1025|       |  // The paper discusses a 4-bit window to optimize 512-bit modular
 1026|       |  // exponentiation, used in RSA-1024 with CRT, but RSA-1024 is no longer
 1027|       |  // important.
 1028|       |  //
 1029|       |  // |bn_mul_mont_gather5| and |bn_power5| implement the "almost" reduction
 1030|       |  // variant, so the values here may not be fully reduced. They are bounded by R
 1031|       |  // (i.e. they fit in |top| words), not |m|. Additionally, we pass these
 1032|       |  // "almost" reduced inputs into |bn_mul_mont|, which implements the normal
 1033|       |  // reduction variant. Given those inputs, |bn_mul_mont| may not give reduced
 1034|       |  // output, but it will still produce "almost" reduced output.
 1035|       |  //
 1036|       |  // TODO(davidben): Using "almost" reduction complicates analysis of this code,
 1037|       |  // and its interaction with other parts of the project. Determine whether this
 1038|       |  // is actually necessary for performance.
 1039|  1.27k|  if (window == 5 && top > 1) {
  ------------------
  |  Branch (1039:7): [True: 304, False: 974]
  |  Branch (1039:22): [True: 255, False: 49]
  ------------------
 1040|       |    // Copy |mont->N| to improve cache locality.
 1041|    255|    BN_ULONG *np = am.d + top;
 1042|  4.03k|    for (i = 0; i < top; i++) {
  ------------------
  |  Branch (1042:17): [True: 3.78k, False: 255]
  ------------------
 1043|  3.78k|      np[i] = mont->N.d[i];
 1044|  3.78k|    }
 1045|       |
 1046|       |    // Fill |powerbuf| with the first 32 powers of |am|.
 1047|    255|    const BN_ULONG *n0 = mont->n0;
 1048|    255|    bn_scatter5(tmp.d, top, powerbuf, 0);
 1049|    255|    bn_scatter5(am.d, am.width, powerbuf, 1);
 1050|    255|    bn_mul_mont(tmp.d, am.d, am.d, np, n0, top);
 1051|    255|    bn_scatter5(tmp.d, top, powerbuf, 2);
 1052|       |
 1053|       |    // Square to compute powers of two.
 1054|  1.02k|    for (i = 4; i < 32; i *= 2) {
  ------------------
  |  Branch (1054:17): [True: 765, False: 255]
  ------------------
 1055|    765|      bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1056|    765|      bn_scatter5(tmp.d, top, powerbuf, i);
 1057|    765|    }
 1058|       |    // Compute odd powers |i| based on |i - 1|, then all powers |i * 2^j|.
 1059|  4.08k|    for (i = 3; i < 32; i += 2) {
  ------------------
  |  Branch (1059:17): [True: 3.82k, False: 255]
  ------------------
 1060|  3.82k|      bn_mul_mont_gather5(tmp.d, am.d, powerbuf, np, n0, top, i - 1);
 1061|  3.82k|      bn_scatter5(tmp.d, top, powerbuf, i);
 1062|  6.63k|      for (int j = 2 * i; j < 32; j *= 2) {
  ------------------
  |  Branch (1062:27): [True: 2.80k, False: 3.82k]
  ------------------
 1063|  2.80k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1064|  2.80k|        bn_scatter5(tmp.d, top, powerbuf, j);
 1065|  2.80k|      }
 1066|  3.82k|    }
 1067|       |
 1068|    255|    bits--;
 1069|  1.24k|    for (wvalue = 0, i = bits % 5; i >= 0; i--, bits--) {
  ------------------
  |  Branch (1069:36): [True: 992, False: 255]
  ------------------
 1070|    992|      wvalue = (wvalue << 1) + BN_is_bit_set(p, bits);
 1071|    992|    }
 1072|    255|    bn_gather5(tmp.d, top, powerbuf, wvalue);
 1073|       |
 1074|       |    // At this point |bits| is 4 mod 5 and at least -1. (|bits| is the first bit
 1075|       |    // that has not been read yet.)
 1076|    255|    assert(bits >= -1 && (bits == -1 || bits % 5 == 4));
 1077|       |
 1078|       |    // Scan the exponent one window at a time starting from the most
 1079|       |    // significant bits.
 1080|    255|    if (top & 7) {
  ------------------
  |  Branch (1080:9): [True: 33, False: 222]
  ------------------
 1081|  5.84k|      while (bits >= 0) {
  ------------------
  |  Branch (1081:14): [True: 5.81k, False: 33]
  ------------------
 1082|  34.8k|        for (wvalue = 0, i = 0; i < 5; i++, bits--) {
  ------------------
  |  Branch (1082:33): [True: 29.0k, False: 5.81k]
  ------------------
 1083|  29.0k|          wvalue = (wvalue << 1) + BN_is_bit_set(p, bits);
 1084|  29.0k|        }
 1085|       |
 1086|  5.81k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1087|  5.81k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1088|  5.81k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1089|  5.81k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1090|  5.81k|        bn_mul_mont(tmp.d, tmp.d, tmp.d, np, n0, top);
 1091|  5.81k|        bn_mul_mont_gather5(tmp.d, tmp.d, powerbuf, np, n0, top, wvalue);
 1092|  5.81k|      }
 1093|    222|    } else {
 1094|    222|      const uint8_t *p_bytes = (const uint8_t *)p->d;
 1095|    222|      assert(bits < max_bits);
 1096|       |      // |p = 0| has been handled as a special case, so |max_bits| is at least
 1097|       |      // one word.
 1098|    222|      assert(max_bits >= 64);
 1099|       |
 1100|       |      // If the first bit to be read lands in the last byte, unroll the first
 1101|       |      // iteration to avoid reading past the bounds of |p->d|. (After the first
 1102|       |      // iteration, we are guaranteed to be past the last byte.) Note |bits|
 1103|       |      // here is the top bit, inclusive.
 1104|    222|      if (bits - 4 >= max_bits - 8) {
  ------------------
  |  Branch (1104:11): [True: 12, False: 210]
  ------------------
 1105|       |        // Read five bits from |bits-4| through |bits|, inclusive.
 1106|     12|        wvalue = p_bytes[p->width * BN_BYTES - 1];
  ------------------
  |  |  152|     12|#define BN_BYTES 8
  ------------------
 1107|     12|        wvalue >>= (bits - 4) & 7;
 1108|     12|        wvalue &= 0x1f;
 1109|     12|        bits -= 5;
 1110|     12|        bn_power5(tmp.d, tmp.d, powerbuf, np, n0, top, wvalue);
 1111|     12|      }
 1112|  46.6k|      while (bits >= 0) {
  ------------------
  |  Branch (1112:14): [True: 46.4k, False: 222]
  ------------------
 1113|       |        // Read five bits from |bits-4| through |bits|, inclusive.
 1114|  46.4k|        int first_bit = bits - 4;
 1115|  46.4k|        uint16_t val;
 1116|  46.4k|        OPENSSL_memcpy(&val, p_bytes + (first_bit >> 3), sizeof(val));
 1117|  46.4k|        val >>= first_bit & 7;
 1118|  46.4k|        val &= 0x1f;
 1119|  46.4k|        bits -= 5;
 1120|  46.4k|        bn_power5(tmp.d, tmp.d, powerbuf, np, n0, top, val);
 1121|  46.4k|      }
 1122|    222|    }
 1123|       |    // The result is now in |tmp| in Montgomery form, but it may not be fully
 1124|       |    // reduced. This is within bounds for |BN_from_montgomery| (tmp < R <= m*R)
 1125|       |    // so it will, when converting from Montgomery form, produce a fully reduced
 1126|       |    // result.
 1127|       |    //
 1128|       |    // This differs from Figure 2 of the paper, which uses AMM(h, 1) to convert
 1129|       |    // from Montgomery form with unreduced output, followed by an extra
 1130|       |    // reduction step. In the paper's terminology, we replace steps 9 and 10
 1131|       |    // with MM(h, 1).
 1132|    255|  } else
 1133|  1.02k|#endif
 1134|  1.02k|  {
 1135|  1.02k|    copy_to_prebuf(&tmp, top, powerbuf, 0, window);
 1136|  1.02k|    copy_to_prebuf(&am, top, powerbuf, 1, window);
 1137|       |
 1138|       |    // If the window size is greater than 1, then calculate
 1139|       |    // val[i=2..2^winsize-1]. Powers are computed as a*a^(i-1)
 1140|       |    // (even powers could instead be computed as (a^(i/2))^2
 1141|       |    // to use the slight performance advantage of sqr over mul).
 1142|  1.02k|    if (window > 1) {
  ------------------
  |  Branch (1142:9): [True: 1.02k, False: 0]
  ------------------
 1143|  1.02k|      if (!BN_mod_mul_montgomery(&tmp, &am, &am, mont, ctx)) {
  ------------------
  |  Branch (1143:11): [True: 0, False: 1.02k]
  ------------------
 1144|      0|        goto err;
 1145|      0|      }
 1146|       |
 1147|  1.02k|      copy_to_prebuf(&tmp, top, powerbuf, 2, window);
 1148|       |
 1149|  7.91k|      for (i = 3; i < num_powers; i++) {
  ------------------
  |  Branch (1149:19): [True: 6.89k, False: 1.02k]
  ------------------
 1150|       |        // Calculate a^i = a^(i-1) * a
 1151|  6.89k|        if (!BN_mod_mul_montgomery(&tmp, &am, &tmp, mont, ctx)) {
  ------------------
  |  Branch (1151:13): [True: 0, False: 6.89k]
  ------------------
 1152|      0|          goto err;
 1153|      0|        }
 1154|       |
 1155|  6.89k|        copy_to_prebuf(&tmp, top, powerbuf, i, window);
 1156|  6.89k|      }
 1157|  1.02k|    }
 1158|       |
 1159|  1.02k|    bits--;
 1160|  2.37k|    for (wvalue = 0, i = bits % window; i >= 0; i--, bits--) {
  ------------------
  |  Branch (1160:41): [True: 1.35k, False: 1.02k]
  ------------------
 1161|  1.35k|      wvalue = (wvalue << 1) + BN_is_bit_set(p, bits);
 1162|  1.35k|    }
 1163|  1.02k|    if (!copy_from_prebuf(&tmp, top, powerbuf, wvalue, window)) {
  ------------------
  |  Branch (1163:9): [True: 0, False: 1.02k]
  ------------------
 1164|      0|      goto err;
 1165|      0|    }
 1166|       |
 1167|       |    // Scan the exponent one window at a time starting from the most
 1168|       |    // significant bits.
 1169|  36.9k|    while (bits >= 0) {
  ------------------
  |  Branch (1169:12): [True: 35.9k, False: 1.02k]
  ------------------
 1170|  35.9k|      wvalue = 0;  // The 'value' of the window
 1171|       |
 1172|       |      // Scan the window, squaring the result as we go
 1173|   175k|      for (i = 0; i < window; i++, bits--) {
  ------------------
  |  Branch (1173:19): [True: 139k, False: 35.9k]
  ------------------
 1174|   139k|        if (!BN_mod_mul_montgomery(&tmp, &tmp, &tmp, mont, ctx)) {
  ------------------
  |  Branch (1174:13): [True: 0, False: 139k]
  ------------------
 1175|      0|          goto err;
 1176|      0|        }
 1177|   139k|        wvalue = (wvalue << 1) + BN_is_bit_set(p, bits);
 1178|   139k|      }
 1179|       |
 1180|       |      // Fetch the appropriate pre-computed value from the pre-buf
 1181|  35.9k|      if (!copy_from_prebuf(&am, top, powerbuf, wvalue, window)) {
  ------------------
  |  Branch (1181:11): [True: 0, False: 35.9k]
  ------------------
 1182|      0|        goto err;
 1183|      0|      }
 1184|       |
 1185|       |      // Multiply the result into the intermediate result
 1186|  35.9k|      if (!BN_mod_mul_montgomery(&tmp, &tmp, &am, mont, ctx)) {
  ------------------
  |  Branch (1186:11): [True: 0, False: 35.9k]
  ------------------
 1187|      0|        goto err;
 1188|      0|      }
 1189|  35.9k|    }
 1190|  1.02k|  }
 1191|       |
 1192|       |  // Convert the final result from Montgomery to standard format. If we used the
 1193|       |  // |OPENSSL_BN_ASM_MONT5| codepath, |tmp| may not be fully reduced. It is only
 1194|       |  // bounded by R rather than |m|. However, that is still within bounds for
 1195|       |  // |BN_from_montgomery|, which implements full Montgomery reduction, not
 1196|       |  // "almost" Montgomery reduction.
 1197|  1.27k|  if (!BN_from_montgomery(rr, &tmp, mont, ctx)) {
  ------------------
  |  Branch (1197:7): [True: 0, False: 1.27k]
  ------------------
 1198|      0|    goto err;
 1199|      0|  }
 1200|  1.27k|  ret = 1;
 1201|       |
 1202|  1.27k|err:
 1203|  1.27k|  BN_MONT_CTX_free(new_mont);
 1204|  1.27k|  if (powerbuf != NULL && powerbuf_free == NULL) {
  ------------------
  |  Branch (1204:7): [True: 1.27k, False: 0]
  |  Branch (1204:27): [True: 1.27k, False: 8]
  ------------------
 1205|  1.27k|    OPENSSL_cleanse(powerbuf, powerbuf_len);
 1206|  1.27k|  }
 1207|  1.27k|  OPENSSL_free(powerbuf_free);
 1208|  1.27k|  return ret;
 1209|  1.27k|}
bcm.c:mod_exp_recp:
  431|  1.50k|                        const BIGNUM *m, BN_CTX *ctx) {
  432|  1.50k|  int i, j, ret = 0, wstart, window;
  433|  1.50k|  int start = 1;
  434|  1.50k|  BIGNUM *aa;
  435|       |  // Table of variables obtained from 'ctx'
  436|  1.50k|  BIGNUM *val[TABLE_SIZE];
  437|  1.50k|  BN_RECP_CTX recp;
  438|       |
  439|       |  // This function is only called on even moduli.
  440|  1.50k|  assert(!BN_is_odd(m));
  441|       |
  442|  1.50k|  int bits = BN_num_bits(p);
  443|  1.50k|  if (bits == 0) {
  ------------------
  |  Branch (443:7): [True: 45, False: 1.45k]
  ------------------
  444|     45|    return BN_one(r);
  445|     45|  }
  446|       |
  447|  1.45k|  BN_RECP_CTX_init(&recp);
  448|  1.45k|  BN_CTX_start(ctx);
  449|  1.45k|  aa = BN_CTX_get(ctx);
  450|  1.45k|  val[0] = BN_CTX_get(ctx);
  451|  1.45k|  if (!aa || !val[0]) {
  ------------------
  |  Branch (451:7): [True: 0, False: 1.45k]
  |  Branch (451:14): [True: 0, False: 1.45k]
  ------------------
  452|      0|    goto err;
  453|      0|  }
  454|       |
  455|  1.45k|  if (m->neg) {
  ------------------
  |  Branch (455:7): [True: 0, False: 1.45k]
  ------------------
  456|       |    // ignore sign of 'm'
  457|      0|    if (!BN_copy(aa, m)) {
  ------------------
  |  Branch (457:9): [True: 0, False: 0]
  ------------------
  458|      0|      goto err;
  459|      0|    }
  460|      0|    aa->neg = 0;
  461|      0|    if (BN_RECP_CTX_set(&recp, aa, ctx) <= 0) {
  ------------------
  |  Branch (461:9): [True: 0, False: 0]
  ------------------
  462|      0|      goto err;
  463|      0|    }
  464|  1.45k|  } else {
  465|  1.45k|    if (BN_RECP_CTX_set(&recp, m, ctx) <= 0) {
  ------------------
  |  Branch (465:9): [True: 0, False: 1.45k]
  ------------------
  466|      0|      goto err;
  467|      0|    }
  468|  1.45k|  }
  469|       |
  470|  1.45k|  if (!BN_nnmod(val[0], a, m, ctx)) {
  ------------------
  |  Branch (470:7): [True: 0, False: 1.45k]
  ------------------
  471|      0|    goto err;  // 1
  472|      0|  }
  473|  1.45k|  if (BN_is_zero(val[0])) {
  ------------------
  |  Branch (473:7): [True: 57, False: 1.39k]
  ------------------
  474|     57|    BN_zero(r);
  475|     57|    ret = 1;
  476|     57|    goto err;
  477|     57|  }
  478|       |
  479|  1.39k|  window = BN_window_bits_for_exponent_size(bits);
  480|  1.39k|  if (window > 1) {
  ------------------
  |  Branch (480:7): [True: 361, False: 1.03k]
  ------------------
  481|    361|    if (!BN_mod_mul_reciprocal(aa, val[0], val[0], &recp, ctx)) {
  ------------------
  |  Branch (481:9): [True: 0, False: 361]
  ------------------
  482|      0|      goto err;  // 2
  483|      0|    }
  484|    361|    j = 1 << (window - 1);
  485|  3.01k|    for (i = 1; i < j; i++) {
  ------------------
  |  Branch (485:17): [True: 2.65k, False: 361]
  ------------------
  486|  2.65k|      if (((val[i] = BN_CTX_get(ctx)) == NULL) ||
  ------------------
  |  Branch (486:11): [True: 0, False: 2.65k]
  ------------------
  487|  2.65k|          !BN_mod_mul_reciprocal(val[i], val[i - 1], aa, &recp, ctx)) {
  ------------------
  |  Branch (487:11): [True: 0, False: 2.65k]
  ------------------
  488|      0|        goto err;
  489|      0|      }
  490|  2.65k|    }
  491|    361|  }
  492|       |
  493|  1.39k|  start = 1;  // This is used to avoid multiplication etc
  494|       |              // when there is only the value '1' in the
  495|       |              // buffer.
  496|  1.39k|  wstart = bits - 1;  // The top bit of the window
  497|       |
  498|  1.39k|  if (!BN_one(r)) {
  ------------------
  |  Branch (498:7): [True: 0, False: 1.39k]
  ------------------
  499|      0|    goto err;
  500|      0|  }
  501|       |
  502|  67.4k|  for (;;) {
  503|  67.4k|    int wvalue;  // The 'value' of the window
  504|  67.4k|    int wend;  // The bottom bit of the window
  505|       |
  506|  67.4k|    if (!BN_is_bit_set(p, wstart)) {
  ------------------
  |  Branch (506:9): [True: 50.0k, False: 17.4k]
  ------------------
  507|  50.0k|      if (!start) {
  ------------------
  |  Branch (507:11): [True: 50.0k, False: 0]
  ------------------
  508|  50.0k|        if (!BN_mod_mul_reciprocal(r, r, r, &recp, ctx)) {
  ------------------
  |  Branch (508:13): [True: 0, False: 50.0k]
  ------------------
  509|      0|          goto err;
  510|      0|        }
  511|  50.0k|      }
  512|  50.0k|      if (wstart == 0) {
  ------------------
  |  Branch (512:11): [True: 409, False: 49.6k]
  ------------------
  513|    409|        break;
  514|    409|      }
  515|  49.6k|      wstart--;
  516|  49.6k|      continue;
  517|  50.0k|    }
  518|       |
  519|       |    // We now have wstart on a 'set' bit, we now need to work out
  520|       |    // how bit a window to do.  To do this we need to scan
  521|       |    // forward until the last set bit before the end of the
  522|       |    // window
  523|  17.4k|    wvalue = 1;
  524|  17.4k|    wend = 0;
  525|  63.0k|    for (i = 1; i < window; i++) {
  ------------------
  |  Branch (525:17): [True: 45.7k, False: 17.2k]
  ------------------
  526|  45.7k|      if (wstart - i < 0) {
  ------------------
  |  Branch (526:11): [True: 165, False: 45.6k]
  ------------------
  527|    165|        break;
  528|    165|      }
  529|  45.6k|      if (BN_is_bit_set(p, wstart - i)) {
  ------------------
  |  Branch (529:11): [True: 28.9k, False: 16.6k]
  ------------------
  530|  28.9k|        wvalue <<= (i - wend);
  531|  28.9k|        wvalue |= 1;
  532|  28.9k|        wend = i;
  533|  28.9k|      }
  534|  45.6k|    }
  535|       |
  536|       |    // wend is the size of the current window
  537|  17.4k|    j = wend + 1;
  538|       |    // add the 'bytes above'
  539|  17.4k|    if (!start) {
  ------------------
  |  Branch (539:9): [True: 16.0k, False: 1.39k]
  ------------------
  540|  65.7k|      for (i = 0; i < j; i++) {
  ------------------
  |  Branch (540:19): [True: 49.7k, False: 16.0k]
  ------------------
  541|  49.7k|        if (!BN_mod_mul_reciprocal(r, r, r, &recp, ctx)) {
  ------------------
  |  Branch (541:13): [True: 0, False: 49.7k]
  ------------------
  542|      0|          goto err;
  543|      0|        }
  544|  49.7k|      }
  545|  16.0k|    }
  546|       |
  547|       |    // wvalue will be an odd number < 2^window
  548|  17.4k|    if (!BN_mod_mul_reciprocal(r, r, val[wvalue >> 1], &recp, ctx)) {
  ------------------
  |  Branch (548:9): [True: 0, False: 17.4k]
  ------------------
  549|      0|      goto err;
  550|      0|    }
  551|       |
  552|       |    // move the 'window' down further
  553|  17.4k|    wstart -= wend + 1;
  554|  17.4k|    start = 0;
  555|  17.4k|    if (wstart < 0) {
  ------------------
  |  Branch (555:9): [True: 990, False: 16.4k]
  ------------------
  556|    990|      break;
  557|    990|    }
  558|  17.4k|  }
  559|  1.39k|  ret = 1;
  560|       |
  561|  1.45k|err:
  562|  1.45k|  BN_CTX_end(ctx);
  563|  1.45k|  BN_RECP_CTX_free(&recp);
  564|  1.45k|  return ret;
  565|  1.39k|}
bcm.c:BN_RECP_CTX_init:
  183|  1.45k|static void BN_RECP_CTX_init(BN_RECP_CTX *recp) {
  184|  1.45k|  BN_init(&recp->N);
  185|  1.45k|  BN_init(&recp->Nr);
  186|  1.45k|  recp->num_bits = 0;
  187|  1.45k|  recp->shift = 0;
  188|  1.45k|  recp->flags = 0;
  189|  1.45k|}
bcm.c:BN_RECP_CTX_set:
  200|  1.45k|static int BN_RECP_CTX_set(BN_RECP_CTX *recp, const BIGNUM *d, BN_CTX *ctx) {
  201|  1.45k|  if (!BN_copy(&(recp->N), d)) {
  ------------------
  |  Branch (201:7): [True: 0, False: 1.45k]
  ------------------
  202|      0|    return 0;
  203|      0|  }
  204|  1.45k|  BN_zero(&recp->Nr);
  205|  1.45k|  recp->num_bits = BN_num_bits(d);
  206|  1.45k|  recp->shift = 0;
  207|       |
  208|  1.45k|  return 1;
  209|  1.45k|}
bcm.c:BN_mod_mul_reciprocal:
  344|   120k|                                 BN_RECP_CTX *recp, BN_CTX *ctx) {
  345|   120k|  int ret = 0;
  346|   120k|  BIGNUM *a;
  347|   120k|  const BIGNUM *ca;
  348|       |
  349|   120k|  BN_CTX_start(ctx);
  350|   120k|  a = BN_CTX_get(ctx);
  351|   120k|  if (a == NULL) {
  ------------------
  |  Branch (351:7): [True: 0, False: 120k]
  ------------------
  352|      0|    goto err;
  353|      0|  }
  354|       |
  355|   120k|  if (y != NULL) {
  ------------------
  |  Branch (355:7): [True: 120k, False: 0]
  ------------------
  356|   120k|    if (x == y) {
  ------------------
  |  Branch (356:9): [True: 100k, False: 20.0k]
  ------------------
  357|   100k|      if (!BN_sqr(a, x, ctx)) {
  ------------------
  |  Branch (357:11): [True: 0, False: 100k]
  ------------------
  358|      0|        goto err;
  359|      0|      }
  360|   100k|    } else {
  361|  20.0k|      if (!BN_mul(a, x, y, ctx)) {
  ------------------
  |  Branch (361:11): [True: 0, False: 20.0k]
  ------------------
  362|      0|        goto err;
  363|      0|      }
  364|  20.0k|    }
  365|   120k|    ca = a;
  366|   120k|  } else {
  367|      0|    ca = x;  // Just do the mod
  368|      0|  }
  369|       |
  370|   120k|  ret = BN_div_recp(NULL, r, ca, recp, ctx);
  371|       |
  372|   120k|err:
  373|   120k|  BN_CTX_end(ctx);
  374|   120k|  return ret;
  375|   120k|}
bcm.c:BN_div_recp:
  241|   120k|                       BN_RECP_CTX *recp, BN_CTX *ctx) {
  242|   120k|  int i, j, ret = 0;
  243|   120k|  BIGNUM *a, *b, *d, *r;
  244|       |
  245|   120k|  BN_CTX_start(ctx);
  246|   120k|  a = BN_CTX_get(ctx);
  247|   120k|  b = BN_CTX_get(ctx);
  248|   120k|  if (dv != NULL) {
  ------------------
  |  Branch (248:7): [True: 0, False: 120k]
  ------------------
  249|      0|    d = dv;
  250|   120k|  } else {
  251|   120k|    d = BN_CTX_get(ctx);
  252|   120k|  }
  253|       |
  254|   120k|  if (rem != NULL) {
  ------------------
  |  Branch (254:7): [True: 120k, False: 0]
  ------------------
  255|   120k|    r = rem;
  256|   120k|  } else {
  257|      0|    r = BN_CTX_get(ctx);
  258|      0|  }
  259|       |
  260|   120k|  if (a == NULL || b == NULL || d == NULL || r == NULL) {
  ------------------
  |  Branch (260:7): [True: 0, False: 120k]
  |  Branch (260:20): [True: 0, False: 120k]
  |  Branch (260:33): [True: 0, False: 120k]
  |  Branch (260:46): [True: 0, False: 120k]
  ------------------
  261|      0|    goto err;
  262|      0|  }
  263|       |
  264|   120k|  if (BN_ucmp(m, &recp->N) < 0) {
  ------------------
  |  Branch (264:7): [True: 41.9k, False: 78.3k]
  ------------------
  265|  41.9k|    BN_zero(d);
  266|  41.9k|    if (!BN_copy(r, m)) {
  ------------------
  |  Branch (266:9): [True: 0, False: 41.9k]
  ------------------
  267|      0|      goto err;
  268|      0|    }
  269|  41.9k|    BN_CTX_end(ctx);
  270|  41.9k|    return 1;
  271|  41.9k|  }
  272|       |
  273|       |  // We want the remainder
  274|       |  // Given input of ABCDEF / ab
  275|       |  // we need multiply ABCDEF by 3 digests of the reciprocal of ab
  276|       |
  277|       |  // i := max(BN_num_bits(m), 2*BN_num_bits(N))
  278|  78.3k|  i = BN_num_bits(m);
  279|  78.3k|  j = recp->num_bits << 1;
  280|  78.3k|  if (j > i) {
  ------------------
  |  Branch (280:7): [True: 71.9k, False: 6.38k]
  ------------------
  281|  71.9k|    i = j;
  282|  71.9k|  }
  283|       |
  284|       |  // Nr := round(2^i / N)
  285|  78.3k|  if (i != recp->shift) {
  ------------------
  |  Branch (285:7): [True: 1.35k, False: 76.9k]
  ------------------
  286|  1.35k|    recp->shift =
  287|  1.35k|        BN_reciprocal(&(recp->Nr), &(recp->N), i,
  288|  1.35k|                      ctx);  // BN_reciprocal returns i, or -1 for an error
  289|  1.35k|  }
  290|       |
  291|  78.3k|  if (recp->shift == -1) {
  ------------------
  |  Branch (291:7): [True: 0, False: 78.3k]
  ------------------
  292|      0|    goto err;
  293|      0|  }
  294|       |
  295|       |  // d := |round(round(m / 2^BN_num_bits(N)) * recp->Nr / 2^(i -
  296|       |  // BN_num_bits(N)))|
  297|       |  //    = |round(round(m / 2^BN_num_bits(N)) * round(2^i / N) / 2^(i -
  298|       |  // BN_num_bits(N)))|
  299|       |  //   <= |(m / 2^BN_num_bits(N)) * (2^i / N) * (2^BN_num_bits(N) / 2^i)|
  300|       |  //    = |m/N|
  301|  78.3k|  if (!BN_rshift(a, m, recp->num_bits)) {
  ------------------
  |  Branch (301:7): [True: 0, False: 78.3k]
  ------------------
  302|      0|    goto err;
  303|      0|  }
  304|  78.3k|  if (!BN_mul(b, a, &(recp->Nr), ctx)) {
  ------------------
  |  Branch (304:7): [True: 0, False: 78.3k]
  ------------------
  305|      0|    goto err;
  306|      0|  }
  307|  78.3k|  if (!BN_rshift(d, b, i - recp->num_bits)) {
  ------------------
  |  Branch (307:7): [True: 0, False: 78.3k]
  ------------------
  308|      0|    goto err;
  309|      0|  }
  310|  78.3k|  d->neg = 0;
  311|       |
  312|  78.3k|  if (!BN_mul(b, &(recp->N), d, ctx)) {
  ------------------
  |  Branch (312:7): [True: 0, False: 78.3k]
  ------------------
  313|      0|    goto err;
  314|      0|  }
  315|  78.3k|  if (!BN_usub(r, m, b)) {
  ------------------
  |  Branch (315:7): [True: 0, False: 78.3k]
  ------------------
  316|      0|    goto err;
  317|      0|  }
  318|  78.3k|  r->neg = 0;
  319|       |
  320|  78.3k|  j = 0;
  321|   132k|  while (BN_ucmp(r, &(recp->N)) >= 0) {
  ------------------
  |  Branch (321:10): [True: 54.5k, False: 78.3k]
  ------------------
  322|  54.5k|    if (j++ > 2) {
  ------------------
  |  Branch (322:9): [True: 0, False: 54.5k]
  ------------------
  323|      0|      OPENSSL_PUT_ERROR(BN, BN_R_BAD_RECIPROCAL);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  324|      0|      goto err;
  325|      0|    }
  326|  54.5k|    if (!BN_usub(r, r, &(recp->N))) {
  ------------------
  |  Branch (326:9): [True: 0, False: 54.5k]
  ------------------
  327|      0|      goto err;
  328|      0|    }
  329|  54.5k|    if (!BN_add_word(d, 1)) {
  ------------------
  |  Branch (329:9): [True: 0, False: 54.5k]
  ------------------
  330|      0|      goto err;
  331|      0|    }
  332|  54.5k|  }
  333|       |
  334|  78.3k|  r->neg = BN_is_zero(r) ? 0 : m->neg;
  ------------------
  |  Branch (334:12): [True: 117, False: 78.1k]
  ------------------
  335|  78.3k|  d->neg = m->neg ^ recp->N.neg;
  336|  78.3k|  ret = 1;
  337|       |
  338|  78.3k|err:
  339|  78.3k|  BN_CTX_end(ctx);
  340|  78.3k|  return ret;
  341|  78.3k|}
bcm.c:BN_reciprocal:
  215|  1.35k|static int BN_reciprocal(BIGNUM *r, const BIGNUM *m, int len, BN_CTX *ctx) {
  216|  1.35k|  int ret = -1;
  217|  1.35k|  BIGNUM *t;
  218|       |
  219|  1.35k|  BN_CTX_start(ctx);
  220|  1.35k|  t = BN_CTX_get(ctx);
  221|  1.35k|  if (t == NULL) {
  ------------------
  |  Branch (221:7): [True: 0, False: 1.35k]
  ------------------
  222|      0|    goto err;
  223|      0|  }
  224|       |
  225|  1.35k|  if (!BN_set_bit(t, len)) {
  ------------------
  |  Branch (225:7): [True: 0, False: 1.35k]
  ------------------
  226|      0|    goto err;
  227|      0|  }
  228|       |
  229|  1.35k|  if (!BN_div(r, NULL, t, m, ctx)) {
  ------------------
  |  Branch (229:7): [True: 0, False: 1.35k]
  ------------------
  230|      0|    goto err;
  231|      0|  }
  232|       |
  233|  1.35k|  ret = len;
  234|       |
  235|  1.35k|err:
  236|  1.35k|  BN_CTX_end(ctx);
  237|  1.35k|  return ret;
  238|  1.35k|}
bcm.c:BN_RECP_CTX_free:
  191|  1.45k|static void BN_RECP_CTX_free(BN_RECP_CTX *recp) {
  192|  1.45k|  if (recp == NULL) {
  ------------------
  |  Branch (192:7): [True: 0, False: 1.45k]
  ------------------
  193|      0|    return;
  194|      0|  }
  195|       |
  196|  1.45k|  BN_free(&recp->N);
  197|  1.45k|  BN_free(&recp->Nr);
  198|  1.45k|}
bcm.c:BN_window_bits_for_exponent_size:
  400|  3.58k|static int BN_window_bits_for_exponent_size(size_t b) {
  401|  3.58k|  if (b > 671) {
  ------------------
  |  Branch (401:7): [True: 532, False: 3.05k]
  ------------------
  402|    532|    return 6;
  403|    532|  }
  404|  3.05k|  if (b > 239) {
  ------------------
  |  Branch (404:7): [True: 137, False: 2.91k]
  ------------------
  405|    137|    return 5;
  406|    137|  }
  407|  2.91k|  if (b > 79) {
  ------------------
  |  Branch (407:7): [True: 118, False: 2.79k]
  ------------------
  408|    118|    return 4;
  409|    118|  }
  410|  2.79k|  if (b > 23) {
  ------------------
  |  Branch (410:7): [True: 702, False: 2.09k]
  ------------------
  411|    702|    return 3;
  412|    702|  }
  413|  2.09k|  return 1;
  414|  2.79k|}
bcm.c:copy_to_prebuf:
  840|  9.96k|                           int window) {
  841|  9.96k|  int ret = bn_copy_words(table + idx * top, top, b);
  842|  9.96k|  assert(ret);  // |b| is guaranteed to fit.
  843|  9.96k|  (void)ret;
  844|  9.96k|}
bcm.c:copy_from_prebuf:
  847|  36.9k|                            int window) {
  848|  36.9k|  if (!bn_wexpand(b, top)) {
  ------------------
  |  Branch (848:7): [True: 0, False: 36.9k]
  ------------------
  849|      0|    return 0;
  850|      0|  }
  851|       |
  852|  36.9k|  OPENSSL_memset(b->d, 0, sizeof(BN_ULONG) * top);
  853|  36.9k|  const int width = 1 << window;
  854|   700k|  for (int i = 0; i < width; i++, table += top) {
  ------------------
  |  Branch (854:19): [True: 663k, False: 36.9k]
  ------------------
  855|       |    // Use a value barrier to prevent Clang from adding a branch when |i != idx|
  856|       |    // and making this copy not constant time. Clang is still allowed to learn
  857|       |    // that |mask| is constant across the inner loop, so this won't inhibit any
  858|       |    // vectorization it might do.
  859|   663k|    BN_ULONG mask = value_barrier_w(constant_time_eq_int(i, idx));
  860|  1.85M|    for (int j = 0; j < top; j++) {
  ------------------
  |  Branch (860:21): [True: 1.19M, False: 663k]
  ------------------
  861|  1.19M|      b->d[j] |= table[j] & mask;
  862|  1.19M|    }
  863|   663k|  }
  864|       |
  865|  36.9k|  b->width = top;
  866|  36.9k|  return 1;
  867|  36.9k|}

BN_MONT_CTX_new:
  124|  2.37k|BN_MONT_CTX *BN_MONT_CTX_new(void) {
  125|  2.37k|  BN_MONT_CTX *ret = OPENSSL_malloc(sizeof(BN_MONT_CTX));
  126|       |
  127|  2.37k|  if (ret == NULL) {
  ------------------
  |  Branch (127:7): [True: 0, False: 2.37k]
  ------------------
  128|      0|    return NULL;
  129|      0|  }
  130|       |
  131|  2.37k|  OPENSSL_memset(ret, 0, sizeof(BN_MONT_CTX));
  132|  2.37k|  BN_init(&ret->RR);
  133|  2.37k|  BN_init(&ret->N);
  134|       |
  135|  2.37k|  return ret;
  136|  2.37k|}
BN_MONT_CTX_free:
  138|  4.74k|void BN_MONT_CTX_free(BN_MONT_CTX *mont) {
  139|  4.74k|  if (mont == NULL) {
  ------------------
  |  Branch (139:7): [True: 2.37k, False: 2.37k]
  ------------------
  140|  2.37k|    return;
  141|  2.37k|  }
  142|       |
  143|  2.37k|  BN_free(&mont->RR);
  144|  2.37k|  BN_free(&mont->N);
  145|  2.37k|  OPENSSL_free(mont);
  146|  2.37k|}
BN_MONT_CTX_set:
  210|  1.27k|int BN_MONT_CTX_set(BN_MONT_CTX *mont, const BIGNUM *mod, BN_CTX *ctx) {
  211|  1.27k|  if (!bn_mont_ctx_set_N_and_n0(mont, mod)) {
  ------------------
  |  Branch (211:7): [True: 0, False: 1.27k]
  ------------------
  212|      0|    return 0;
  213|      0|  }
  214|       |
  215|  1.27k|  BN_CTX *new_ctx = NULL;
  216|  1.27k|  if (ctx == NULL) {
  ------------------
  |  Branch (216:7): [True: 0, False: 1.27k]
  ------------------
  217|      0|    new_ctx = BN_CTX_new();
  218|      0|    if (new_ctx == NULL) {
  ------------------
  |  Branch (218:9): [True: 0, False: 0]
  ------------------
  219|      0|      return 0;
  220|      0|    }
  221|      0|    ctx = new_ctx;
  222|      0|  }
  223|       |
  224|       |  // Save RR = R**2 (mod N). R is the smallest power of 2**BN_BITS2 such that R
  225|       |  // > mod. Even though the assembly on some 32-bit platforms works with 64-bit
  226|       |  // values, using |BN_BITS2| here, rather than |BN_MONT_CTX_N0_LIMBS *
  227|       |  // BN_BITS2|, is correct because R**2 will still be a multiple of the latter
  228|       |  // as |BN_MONT_CTX_N0_LIMBS| is either one or two.
  229|  1.27k|  unsigned lgBigR = mont->N.width * BN_BITS2;
  ------------------
  |  |  151|  1.27k|#define BN_BITS2 64
  ------------------
  230|  1.27k|  BN_zero(&mont->RR);
  231|  1.27k|  int ok = BN_set_bit(&mont->RR, lgBigR * 2) &&
  ------------------
  |  Branch (231:12): [True: 1.27k, False: 0]
  ------------------
  232|  1.27k|           BN_mod(&mont->RR, &mont->RR, &mont->N, ctx) &&
  ------------------
  |  |  547|  2.55k|  BN_div(NULL, (rem), (numerator), (divisor), (ctx))
  |  |  ------------------
  |  |  |  Branch (547:3): [True: 1.27k, False: 0]
  |  |  ------------------
  ------------------
  233|  1.27k|           bn_resize_words(&mont->RR, mont->N.width);
  ------------------
  |  Branch (233:12): [True: 1.27k, False: 0]
  ------------------
  234|  1.27k|  BN_CTX_free(new_ctx);
  235|  1.27k|  return ok;
  236|  1.27k|}
BN_MONT_CTX_new_for_modulus:
  238|  1.27k|BN_MONT_CTX *BN_MONT_CTX_new_for_modulus(const BIGNUM *mod, BN_CTX *ctx) {
  239|  1.27k|  BN_MONT_CTX *mont = BN_MONT_CTX_new();
  240|  1.27k|  if (mont == NULL ||
  ------------------
  |  Branch (240:7): [True: 0, False: 1.27k]
  ------------------
  241|  1.27k|      !BN_MONT_CTX_set(mont, mod, ctx)) {
  ------------------
  |  Branch (241:7): [True: 0, False: 1.27k]
  ------------------
  242|      0|    BN_MONT_CTX_free(mont);
  243|      0|    return NULL;
  244|      0|  }
  245|  1.27k|  return mont;
  246|  1.27k|}
BN_MONT_CTX_new_consttime:
  248|  1.09k|BN_MONT_CTX *BN_MONT_CTX_new_consttime(const BIGNUM *mod, BN_CTX *ctx) {
  249|  1.09k|  BN_MONT_CTX *mont = BN_MONT_CTX_new();
  250|  1.09k|  if (mont == NULL ||
  ------------------
  |  Branch (250:7): [True: 0, False: 1.09k]
  ------------------
  251|  1.09k|      !bn_mont_ctx_set_N_and_n0(mont, mod)) {
  ------------------
  |  Branch (251:7): [True: 0, False: 1.09k]
  ------------------
  252|      0|    goto err;
  253|      0|  }
  254|  1.09k|  unsigned lgBigR = mont->N.width * BN_BITS2;
  ------------------
  |  |  151|  1.09k|#define BN_BITS2 64
  ------------------
  255|  1.09k|  if (!bn_mod_exp_base_2_consttime(&mont->RR, lgBigR * 2, &mont->N, ctx) ||
  ------------------
  |  Branch (255:7): [True: 0, False: 1.09k]
  ------------------
  256|  1.09k|      !bn_resize_words(&mont->RR, mont->N.width)) {
  ------------------
  |  Branch (256:7): [True: 0, False: 1.09k]
  ------------------
  257|      0|    goto err;
  258|      0|  }
  259|  1.09k|  return mont;
  260|       |
  261|      0|err:
  262|      0|  BN_MONT_CTX_free(mont);
  263|      0|  return NULL;
  264|  1.09k|}
BN_to_montgomery:
  286|  3.46k|                     BN_CTX *ctx) {
  287|  3.46k|  return BN_mod_mul_montgomery(ret, a, &mont->RR, mont, ctx);
  288|  3.46k|}
BN_from_montgomery:
  346|  4.47k|                       BN_CTX *ctx) {
  347|  4.47k|  int ret = 0;
  348|  4.47k|  BIGNUM *t;
  349|       |
  350|  4.47k|  BN_CTX_start(ctx);
  351|  4.47k|  t = BN_CTX_get(ctx);
  352|  4.47k|  if (t == NULL ||
  ------------------
  |  Branch (352:7): [True: 0, False: 4.47k]
  ------------------
  353|  4.47k|      !BN_copy(t, a)) {
  ------------------
  |  Branch (353:7): [True: 0, False: 4.47k]
  ------------------
  354|      0|    goto err;
  355|      0|  }
  356|       |
  357|  4.47k|  ret = BN_from_montgomery_word(r, t, mont);
  358|       |
  359|  4.47k|err:
  360|  4.47k|  BN_CTX_end(ctx);
  361|       |
  362|  4.47k|  return ret;
  363|  4.47k|}
bn_one_to_montgomery:
  365|  1.27k|int bn_one_to_montgomery(BIGNUM *r, const BN_MONT_CTX *mont, BN_CTX *ctx) {
  366|       |  // If the high bit of |n| is set, R = 2^(width*BN_BITS2) < 2 * |n|, so we
  367|       |  // compute R - |n| rather than perform Montgomery reduction.
  368|  1.27k|  const BIGNUM *n = &mont->N;
  369|  1.27k|  if (n->width > 0 && (n->d[n->width - 1] >> (BN_BITS2 - 1)) != 0) {
  ------------------
  |  |  151|  1.27k|#define BN_BITS2 64
  ------------------
  |  Branch (369:7): [True: 1.27k, False: 0]
  |  Branch (369:23): [True: 262, False: 1.01k]
  ------------------
  370|    262|    if (!bn_wexpand(r, n->width)) {
  ------------------
  |  Branch (370:9): [True: 0, False: 262]
  ------------------
  371|      0|      return 0;
  372|      0|    }
  373|    262|    r->d[0] = 0 - n->d[0];
  374|  1.70k|    for (int i = 1; i < n->width; i++) {
  ------------------
  |  Branch (374:21): [True: 1.44k, False: 262]
  ------------------
  375|  1.44k|      r->d[i] = ~n->d[i];
  376|  1.44k|    }
  377|    262|    r->width = n->width;
  378|    262|    r->neg = 0;
  379|    262|    return 1;
  380|    262|  }
  381|       |
  382|  1.01k|  return BN_from_montgomery(r, &mont->RR, mont, ctx);
  383|  1.27k|}
BN_mod_mul_montgomery:
  420|   930k|                          const BN_MONT_CTX *mont, BN_CTX *ctx) {
  421|   930k|  if (a->neg || b->neg) {
  ------------------
  |  Branch (421:7): [True: 0, False: 930k]
  |  Branch (421:17): [True: 0, False: 930k]
  ------------------
  422|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  423|      0|    return 0;
  424|      0|  }
  425|       |
  426|   930k|#if defined(OPENSSL_BN_ASM_MONT)
  427|       |  // |bn_mul_mont| requires at least 128 bits of limbs, at least for x86.
  428|   930k|  int num = mont->N.width;
  429|   930k|  if (num >= (128 / BN_BITS2) &&
  ------------------
  |  |  151|   930k|#define BN_BITS2 64
  ------------------
  |  Branch (429:7): [True: 578k, False: 351k]
  ------------------
  430|   930k|      a->width == num &&
  ------------------
  |  Branch (430:7): [True: 577k, False: 256]
  ------------------
  431|   930k|      b->width == num) {
  ------------------
  |  Branch (431:7): [True: 577k, False: 0]
  ------------------
  432|   577k|    if (!bn_wexpand(r, num)) {
  ------------------
  |  Branch (432:9): [True: 0, False: 577k]
  ------------------
  433|      0|      return 0;
  434|      0|    }
  435|       |    // This bound is implied by |bn_mont_ctx_set_N_and_n0|. |bn_mul_mont|
  436|       |    // allocates |num| words on the stack, so |num| cannot be too large.
  437|   577k|    assert((size_t)num <= BN_MONTGOMERY_MAX_WORDS);
  438|   577k|    if (!bn_mul_mont(r->d, a->d, b->d, mont->N.d, mont->n0, num)) {
  ------------------
  |  Branch (438:9): [True: 0, False: 577k]
  ------------------
  439|       |      // The check above ensures this won't happen.
  440|      0|      assert(0);
  441|      0|      OPENSSL_PUT_ERROR(BN, ERR_R_INTERNAL_ERROR);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  442|      0|      return 0;
  443|      0|    }
  444|   577k|    r->neg = 0;
  445|   577k|    r->width = num;
  446|   577k|    return 1;
  447|   577k|  }
  448|   352k|#endif
  449|       |
  450|   352k|  return bn_mod_mul_montgomery_fallback(r, a, b, mont, ctx);
  451|   930k|}
bcm.c:bn_mont_ctx_set_N_and_n0:
  162|  2.37k|static int bn_mont_ctx_set_N_and_n0(BN_MONT_CTX *mont, const BIGNUM *mod) {
  163|  2.37k|  if (BN_is_zero(mod)) {
  ------------------
  |  Branch (163:7): [True: 0, False: 2.37k]
  ------------------
  164|      0|    OPENSSL_PUT_ERROR(BN, BN_R_DIV_BY_ZERO);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  165|      0|    return 0;
  166|      0|  }
  167|  2.37k|  if (!BN_is_odd(mod)) {
  ------------------
  |  Branch (167:7): [True: 0, False: 2.37k]
  ------------------
  168|      0|    OPENSSL_PUT_ERROR(BN, BN_R_CALLED_WITH_EVEN_MODULUS);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  169|      0|    return 0;
  170|      0|  }
  171|  2.37k|  if (BN_is_negative(mod)) {
  ------------------
  |  Branch (171:7): [True: 0, False: 2.37k]
  ------------------
  172|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  173|      0|    return 0;
  174|      0|  }
  175|  2.37k|  if (!bn_fits_in_words(mod, BN_MONTGOMERY_MAX_WORDS)) {
  ------------------
  |  |  363|  2.37k|#define BN_MONTGOMERY_MAX_WORDS (8 * 1024 / sizeof(BN_ULONG))
  ------------------
  |  Branch (175:7): [True: 0, False: 2.37k]
  ------------------
  176|      0|    OPENSSL_PUT_ERROR(BN, BN_R_BIGNUM_TOO_LONG);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  177|      0|    return 0;
  178|      0|  }
  179|       |
  180|       |  // Save the modulus.
  181|  2.37k|  if (!BN_copy(&mont->N, mod)) {
  ------------------
  |  Branch (181:7): [True: 0, False: 2.37k]
  ------------------
  182|      0|    OPENSSL_PUT_ERROR(BN, ERR_R_INTERNAL_ERROR);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  183|      0|    return 0;
  184|      0|  }
  185|       |  // |mont->N| is always stored minimally. Computing RR efficiently leaks the
  186|       |  // size of the modulus. While the modulus may be private in RSA (one of the
  187|       |  // primes), their sizes are public, so this is fine.
  188|  2.37k|  bn_set_minimal_width(&mont->N);
  189|       |
  190|       |  // Find n0 such that n0 * N == -1 (mod r).
  191|       |  //
  192|       |  // Only certain BN_BITS2<=32 platforms actually make use of n0[1]. For the
  193|       |  // others, we could use a shorter R value and use faster |BN_ULONG|-based
  194|       |  // math instead of |uint64_t|-based math, which would be double-precision.
  195|       |  // However, currently only the assembler files know which is which.
  196|  2.37k|  static_assert(BN_MONT_CTX_N0_LIMBS == 1 || BN_MONT_CTX_N0_LIMBS == 2,
  197|  2.37k|                "BN_MONT_CTX_N0_LIMBS value is invalid");
  198|  2.37k|  static_assert(sizeof(BN_ULONG) * BN_MONT_CTX_N0_LIMBS == sizeof(uint64_t),
  199|  2.37k|                "uint64_t is insufficient precision for n0");
  200|  2.37k|  uint64_t n0 = bn_mont_n0(&mont->N);
  201|  2.37k|  mont->n0[0] = (BN_ULONG)n0;
  202|       |#if BN_MONT_CTX_N0_LIMBS == 2
  203|       |  mont->n0[1] = (BN_ULONG)(n0 >> BN_BITS2);
  204|       |#else
  205|  2.37k|  mont->n0[1] = 0;
  206|  2.37k|#endif
  207|  2.37k|  return 1;
  208|  2.37k|}
bcm.c:BN_from_montgomery_word:
  322|   356k|                                   const BN_MONT_CTX *mont) {
  323|   356k|  if (r->neg) {
  ------------------
  |  Branch (323:7): [True: 0, False: 356k]
  ------------------
  324|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  325|      0|    return 0;
  326|      0|  }
  327|       |
  328|   356k|  const BIGNUM *n = &mont->N;
  329|   356k|  if (n->width == 0) {
  ------------------
  |  Branch (329:7): [True: 0, False: 356k]
  ------------------
  330|      0|    ret->width = 0;
  331|      0|    return 1;
  332|      0|  }
  333|       |
  334|   356k|  int max = 2 * n->width;  // carry is stored separately
  335|   356k|  if (!bn_resize_words(r, max) ||
  ------------------
  |  Branch (335:7): [True: 0, False: 356k]
  ------------------
  336|   356k|      !bn_wexpand(ret, n->width)) {
  ------------------
  |  Branch (336:7): [True: 0, False: 356k]
  ------------------
  337|      0|    return 0;
  338|      0|  }
  339|       |
  340|   356k|  ret->width = n->width;
  341|   356k|  ret->neg = 0;
  342|   356k|  return bn_from_montgomery_in_place(ret->d, ret->width, r->d, r->width, mont);
  343|   356k|}
bcm.c:bn_mod_mul_montgomery_fallback:
  388|   352k|                                          BN_CTX *ctx) {
  389|   352k|  int ret = 0;
  390|       |
  391|   352k|  BN_CTX_start(ctx);
  392|   352k|  BIGNUM *tmp = BN_CTX_get(ctx);
  393|   352k|  if (tmp == NULL) {
  ------------------
  |  Branch (393:7): [True: 0, False: 352k]
  ------------------
  394|      0|    goto err;
  395|      0|  }
  396|       |
  397|   352k|  if (a == b) {
  ------------------
  |  Branch (397:7): [True: 288k, False: 63.4k]
  ------------------
  398|   288k|    if (!bn_sqr_consttime(tmp, a, ctx)) {
  ------------------
  |  Branch (398:9): [True: 0, False: 288k]
  ------------------
  399|      0|      goto err;
  400|      0|    }
  401|   288k|  } else {
  402|  63.4k|    if (!bn_mul_consttime(tmp, a, b, ctx)) {
  ------------------
  |  Branch (402:9): [True: 0, False: 63.4k]
  ------------------
  403|      0|      goto err;
  404|      0|    }
  405|  63.4k|  }
  406|       |
  407|       |  // reduce from aRR to aR
  408|   352k|  if (!BN_from_montgomery_word(r, tmp, mont)) {
  ------------------
  |  Branch (408:7): [True: 0, False: 352k]
  ------------------
  409|      0|    goto err;
  410|      0|  }
  411|       |
  412|   352k|  ret = 1;
  413|       |
  414|   352k|err:
  415|   352k|  BN_CTX_end(ctx);
  416|   352k|  return ret;
  417|   352k|}
bcm.c:bn_from_montgomery_in_place:
  291|   356k|                                       size_t num_a, const BN_MONT_CTX *mont) {
  292|   356k|  const BN_ULONG *n = mont->N.d;
  293|   356k|  size_t num_n = mont->N.width;
  294|   356k|  if (num_r != num_n || num_a != 2 * num_n) {
  ------------------
  |  Branch (294:7): [True: 0, False: 356k]
  |  Branch (294:25): [True: 0, False: 356k]
  ------------------
  295|      0|    OPENSSL_PUT_ERROR(BN, ERR_R_SHOULD_NOT_HAVE_BEEN_CALLED);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  296|      0|    return 0;
  297|      0|  }
  298|       |
  299|       |  // Add multiples of |n| to |r| until R = 2^(nl * BN_BITS2) divides it. On
  300|       |  // input, we had |r| < |n| * R, so now |r| < 2 * |n| * R. Note that |r|
  301|       |  // includes |carry| which is stored separately.
  302|   356k|  BN_ULONG n0 = mont->n0[0];
  303|   356k|  BN_ULONG carry = 0;
  304|   737k|  for (size_t i = 0; i < num_n; i++) {
  ------------------
  |  Branch (304:22): [True: 381k, False: 356k]
  ------------------
  305|   381k|    BN_ULONG v = bn_mul_add_words(a + i, n, num_n, a[i] * n0);
  306|   381k|    v += carry + a[i + num_n];
  307|   381k|    carry |= (v != a[i + num_n]);
  308|   381k|    carry &= (v <= a[i + num_n]);
  309|   381k|    a[i + num_n] = v;
  310|   381k|  }
  311|       |
  312|       |  // Shift |num_n| words to divide by R. We have |a| < 2 * |n|. Note that |a|
  313|       |  // includes |carry| which is stored separately.
  314|   356k|  a += num_n;
  315|       |
  316|       |  // |a| thus requires at most one additional subtraction |n| to be reduced.
  317|   356k|  bn_reduce_once(r, a, carry, n, num_n);
  318|   356k|  return 1;
  319|   356k|}

bn_mont_n0:
   33|  2.37k|uint64_t bn_mont_n0(const BIGNUM *n) {
   34|       |  // These conditions are checked by the caller, |BN_MONT_CTX_set| or
   35|       |  // |BN_MONT_CTX_new_consttime|.
   36|  2.37k|  assert(!BN_is_zero(n));
   37|  2.37k|  assert(!BN_is_negative(n));
   38|  2.37k|  assert(BN_is_odd(n));
   39|       |
   40|       |  // r == 2**(BN_MONT_CTX_N0_LIMBS * BN_BITS2) and LG_LITTLE_R == lg(r). This
   41|       |  // ensures that we can do integer division by |r| by simply ignoring
   42|       |  // |BN_MONT_CTX_N0_LIMBS| limbs. Similarly, we can calculate values modulo
   43|       |  // |r| by just looking at the lowest |BN_MONT_CTX_N0_LIMBS| limbs. This is
   44|       |  // what makes Montgomery multiplication efficient.
   45|       |  //
   46|       |  // As shown in Algorithm 1 of "Fast Prime Field Elliptic Curve Cryptography
   47|       |  // with 256 Bit Primes" by Shay Gueron and Vlad Krasnov, in the loop of a
   48|       |  // multi-limb Montgomery multiplication of |a * b (mod n)|, given the
   49|       |  // unreduced product |t == a * b|, we repeatedly calculate:
   50|       |  //
   51|       |  //    t1 := t % r         |t1| is |t|'s lowest limb (see previous paragraph).
   52|       |  //    t2 := t1*n0*n
   53|       |  //    t3 := t + t2
   54|       |  //    t := t3 / r         copy all limbs of |t3| except the lowest to |t|.
   55|       |  //
   56|       |  // In the last step, it would only make sense to ignore the lowest limb of
   57|       |  // |t3| if it were zero. The middle steps ensure that this is the case:
   58|       |  //
   59|       |  //                            t3 ==  0 (mod r)
   60|       |  //                        t + t2 ==  0 (mod r)
   61|       |  //                   t + t1*n0*n ==  0 (mod r)
   62|       |  //                       t1*n0*n == -t (mod r)
   63|       |  //                        t*n0*n == -t (mod r)
   64|       |  //                          n0*n == -1 (mod r)
   65|       |  //                            n0 == -1/n (mod r)
   66|       |  //
   67|       |  // Thus, in each iteration of the loop, we multiply by the constant factor
   68|       |  // |n0|, the negative inverse of n (mod r).
   69|       |
   70|       |  // n_mod_r = n % r. As explained above, this is done by taking the lowest
   71|       |  // |BN_MONT_CTX_N0_LIMBS| limbs of |n|.
   72|  2.37k|  uint64_t n_mod_r = n->d[0];
   73|       |#if BN_MONT_CTX_N0_LIMBS == 2
   74|       |  if (n->width > 1) {
   75|       |    n_mod_r |= (uint64_t)n->d[1] << BN_BITS2;
   76|       |  }
   77|       |#endif
   78|       |
   79|  2.37k|  return bn_neg_inv_mod_r_u64(n_mod_r);
   80|  2.37k|}
bn_mod_exp_base_2_consttime:
  163|  1.09k|                                BN_CTX *ctx) {
  164|  1.09k|  assert(!BN_is_zero(n));
  165|  1.09k|  assert(!BN_is_negative(n));
  166|  1.09k|  assert(BN_is_odd(n));
  167|       |
  168|  1.09k|  BN_zero(r);
  169|       |
  170|  1.09k|  unsigned n_bits = BN_num_bits(n);
  171|  1.09k|  assert(n_bits != 0);
  172|  1.09k|  assert(p > n_bits);
  173|  1.09k|  if (n_bits == 1) {
  ------------------
  |  Branch (173:7): [True: 30, False: 1.06k]
  ------------------
  174|     30|    return 1;
  175|     30|  }
  176|       |
  177|       |  // Set |r| to the larger power of two smaller than |n|, then shift with
  178|       |  // reductions the rest of the way.
  179|  1.06k|  if (!BN_set_bit(r, n_bits - 1) ||
  ------------------
  |  Branch (179:7): [True: 0, False: 1.06k]
  ------------------
  180|  1.06k|      !bn_mod_lshift_consttime(r, r, p - (n_bits - 1), n, ctx)) {
  ------------------
  |  Branch (180:7): [True: 0, False: 1.06k]
  ------------------
  181|      0|    return 0;
  182|      0|  }
  183|       |
  184|  1.06k|  return 1;
  185|  1.06k|}
bcm.c:bn_neg_inv_mod_r_u64:
  104|  2.37k|static uint64_t bn_neg_inv_mod_r_u64(uint64_t n) {
  105|  2.37k|  assert(n % 2 == 1);
  106|       |
  107|       |  // alpha == 2**(lg r - 1) == r / 2.
  108|  2.37k|  static const uint64_t alpha = UINT64_C(1) << (LG_LITTLE_R - 1);
  ------------------
  |  |   31|  2.37k|#define LG_LITTLE_R (BN_MONT_CTX_N0_LIMBS * BN_BITS2)
  |  |  ------------------
  |  |  |  |  158|  2.37k|#define BN_MONT_CTX_N0_LIMBS 1
  |  |  ------------------
  |  |               #define LG_LITTLE_R (BN_MONT_CTX_N0_LIMBS * BN_BITS2)
  |  |  ------------------
  |  |  |  |  151|  2.37k|#define BN_BITS2 64
  |  |  ------------------
  ------------------
  109|       |
  110|  2.37k|  const uint64_t beta = n;
  111|       |
  112|  2.37k|  uint64_t u = 1;
  113|  2.37k|  uint64_t v = 0;
  114|       |
  115|       |  // The invariant maintained from here on is:
  116|       |  // 2**(lg r - i) == u*2*alpha - v*beta.
  117|   154k|  for (size_t i = 0; i < LG_LITTLE_R; ++i) {
  ------------------
  |  |   31|   154k|#define LG_LITTLE_R (BN_MONT_CTX_N0_LIMBS * BN_BITS2)
  |  |  ------------------
  |  |  |  |  158|   154k|#define BN_MONT_CTX_N0_LIMBS 1
  |  |  ------------------
  |  |               #define LG_LITTLE_R (BN_MONT_CTX_N0_LIMBS * BN_BITS2)
  |  |  ------------------
  |  |  |  |  151|   154k|#define BN_BITS2 64
  |  |  ------------------
  ------------------
  |  Branch (117:22): [True: 151k, False: 2.37k]
  ------------------
  118|   151k|#if BN_BITS2 == 64 && defined(BN_ULLONG)
  119|   151k|    assert((BN_ULLONG)(1) << (LG_LITTLE_R - i) ==
  120|   151k|           ((BN_ULLONG)u * 2 * alpha) - ((BN_ULLONG)v * beta));
  121|   151k|#endif
  122|       |
  123|       |    // Delete a common factor of 2 in u and v if |u| is even. Otherwise, set
  124|       |    // |u = (u + beta) / 2| and |v = (v / 2) + alpha|.
  125|       |
  126|   151k|    uint64_t u_is_odd = UINT64_C(0) - (u & 1);  // Either 0xff..ff or 0.
  127|       |
  128|       |    // The addition can overflow, so use Dietz's method for it.
  129|       |    //
  130|       |    // Dietz calculates (x+y)/2 by (x⊕y)>>1 + x&y. This is valid for all
  131|       |    // (unsigned) x and y, even when x+y overflows. Evidence for 32-bit values
  132|       |    // (embedded in 64 bits to so that overflow can be ignored):
  133|       |    //
  134|       |    // (declare-fun x () (_ BitVec 64))
  135|       |    // (declare-fun y () (_ BitVec 64))
  136|       |    // (assert (let (
  137|       |    //    (one (_ bv1 64))
  138|       |    //    (thirtyTwo (_ bv32 64)))
  139|       |    //    (and
  140|       |    //      (bvult x (bvshl one thirtyTwo))
  141|       |    //      (bvult y (bvshl one thirtyTwo))
  142|       |    //      (not (=
  143|       |    //        (bvadd (bvlshr (bvxor x y) one) (bvand x y))
  144|       |    //        (bvlshr (bvadd x y) one)))
  145|       |    // )))
  146|       |    // (check-sat)
  147|   151k|    uint64_t beta_if_u_is_odd = beta & u_is_odd;  // Either |beta| or 0.
  148|   151k|    u = ((u ^ beta_if_u_is_odd) >> 1) + (u & beta_if_u_is_odd);
  149|       |
  150|   151k|    uint64_t alpha_if_u_is_odd = alpha & u_is_odd;  // Either |alpha| or 0.
  151|   151k|    v = (v >> 1) + alpha_if_u_is_odd;
  152|   151k|  }
  153|       |
  154|       |  // The invariant now shows that u*r - v*n == 1 since r == 2 * alpha.
  155|  2.37k|#if BN_BITS2 == 64 && defined(BN_ULLONG)
  156|  2.37k|  assert(1 == ((BN_ULLONG)u * 2 * alpha) - ((BN_ULLONG)v * beta));
  157|  2.37k|#endif
  158|       |
  159|  2.37k|  return v;
  160|  2.37k|}

BN_mul:
  515|   790k|int BN_mul(BIGNUM *r, const BIGNUM *a, const BIGNUM *b, BN_CTX *ctx) {
  516|   790k|  if (!bn_mul_impl(r, a, b, ctx)) {
  ------------------
  |  Branch (516:7): [True: 0, False: 790k]
  ------------------
  517|      0|    return 0;
  518|      0|  }
  519|       |
  520|       |  // This additionally fixes any negative zeros created by |bn_mul_impl|.
  521|   790k|  bn_set_minimal_width(r);
  522|   790k|  return 1;
  523|   790k|}
bn_mul_consttime:
  525|  63.4k|int bn_mul_consttime(BIGNUM *r, const BIGNUM *a, const BIGNUM *b, BN_CTX *ctx) {
  526|       |  // Prevent negative zeros.
  527|  63.4k|  if (a->neg || b->neg) {
  ------------------
  |  Branch (527:7): [True: 0, False: 63.4k]
  |  Branch (527:17): [True: 0, False: 63.4k]
  ------------------
  528|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  529|      0|    return 0;
  530|      0|  }
  531|       |
  532|  63.4k|  return bn_mul_impl(r, a, b, ctx);
  533|  63.4k|}
bn_sqr_consttime:
  668|   388k|int bn_sqr_consttime(BIGNUM *r, const BIGNUM *a, BN_CTX *ctx) {
  669|   388k|  int al = a->width;
  670|   388k|  if (al <= 0) {
  ------------------
  |  Branch (670:7): [True: 11.1k, False: 377k]
  ------------------
  671|  11.1k|    r->width = 0;
  672|  11.1k|    r->neg = 0;
  673|  11.1k|    return 1;
  674|  11.1k|  }
  675|       |
  676|   377k|  int ret = 0;
  677|   377k|  BN_CTX_start(ctx);
  678|   377k|  BIGNUM *rr = (a != r) ? r : BN_CTX_get(ctx);
  ------------------
  |  Branch (678:16): [True: 377k, False: 0]
  ------------------
  679|   377k|  BIGNUM *tmp = BN_CTX_get(ctx);
  680|   377k|  if (!rr || !tmp) {
  ------------------
  |  Branch (680:7): [True: 0, False: 377k]
  |  Branch (680:14): [True: 0, False: 377k]
  ------------------
  681|      0|    goto err;
  682|      0|  }
  683|       |
  684|   377k|  int max = 2 * al;  // Non-zero (from above)
  685|   377k|  if (!bn_wexpand(rr, max)) {
  ------------------
  |  Branch (685:7): [True: 0, False: 377k]
  ------------------
  686|      0|    goto err;
  687|      0|  }
  688|       |
  689|   377k|  if (al == 4) {
  ------------------
  |  Branch (689:7): [True: 348, False: 377k]
  ------------------
  690|    348|    bn_sqr_comba4(rr->d, a->d);
  691|   377k|  } else if (al == 8) {
  ------------------
  |  Branch (691:14): [True: 2.26k, False: 375k]
  ------------------
  692|  2.26k|    bn_sqr_comba8(rr->d, a->d);
  693|   375k|  } else {
  694|   375k|    if (al < BN_SQR_RECURSIVE_SIZE_NORMAL) {
  ------------------
  |  |   71|   375k|#define BN_SQR_RECURSIVE_SIZE_NORMAL BN_MUL_RECURSIVE_SIZE_NORMAL
  |  |  ------------------
  |  |  |  |   70|   375k|#define BN_MUL_RECURSIVE_SIZE_NORMAL 16
  |  |  ------------------
  ------------------
  |  Branch (694:9): [True: 356k, False: 18.4k]
  ------------------
  695|   356k|      BN_ULONG t[BN_SQR_RECURSIVE_SIZE_NORMAL * 2];
  696|   356k|      bn_sqr_normal(rr->d, a->d, al, t);
  697|   356k|    } else {
  698|       |      // If |al| is a power of two, we can use |bn_sqr_recursive|.
  699|  18.4k|      if (al != 0 && (al & (al - 1)) == 0) {
  ------------------
  |  Branch (699:11): [True: 18.4k, False: 0]
  |  Branch (699:22): [True: 16.4k, False: 2.03k]
  ------------------
  700|  16.4k|        if (!bn_wexpand(tmp, al * 4)) {
  ------------------
  |  Branch (700:13): [True: 0, False: 16.4k]
  ------------------
  701|      0|          goto err;
  702|      0|        }
  703|  16.4k|        bn_sqr_recursive(rr->d, a->d, al, tmp->d);
  704|  16.4k|      } else {
  705|  2.03k|        if (!bn_wexpand(tmp, max)) {
  ------------------
  |  Branch (705:13): [True: 0, False: 2.03k]
  ------------------
  706|      0|          goto err;
  707|      0|        }
  708|  2.03k|        bn_sqr_normal(rr->d, a->d, al, tmp->d);
  709|  2.03k|      }
  710|  18.4k|    }
  711|   375k|  }
  712|       |
  713|   377k|  rr->neg = 0;
  714|   377k|  rr->width = max;
  715|       |
  716|   377k|  if (rr != r && !BN_copy(r, rr)) {
  ------------------
  |  Branch (716:7): [True: 0, False: 377k]
  |  Branch (716:18): [True: 0, False: 0]
  ------------------
  717|      0|    goto err;
  718|      0|  }
  719|   377k|  ret = 1;
  720|       |
  721|   377k|err:
  722|   377k|  BN_CTX_end(ctx);
  723|   377k|  return ret;
  724|   377k|}
BN_sqr:
  726|   100k|int BN_sqr(BIGNUM *r, const BIGNUM *a, BN_CTX *ctx) {
  727|   100k|  if (!bn_sqr_consttime(r, a, ctx)) {
  ------------------
  |  Branch (727:7): [True: 0, False: 100k]
  ------------------
  728|      0|    return 0;
  729|      0|  }
  730|       |
  731|   100k|  bn_set_minimal_width(r);
  732|   100k|  return 1;
  733|   100k|}
bcm.c:bn_abs_sub_part_words:
  172|  1.36M|                                      BN_ULONG *tmp) {
  173|  1.36M|  BN_ULONG borrow = bn_sub_part_words(tmp, a, b, cl, dl);
  174|  1.36M|  bn_sub_part_words(r, b, a, cl, -dl);
  175|  1.36M|  int r_len = cl + (dl < 0 ? -dl : dl);
  ------------------
  |  Branch (175:21): [True: 23.6k, False: 1.33M]
  ------------------
  176|  1.36M|  borrow = 0 - borrow;
  177|  1.36M|  bn_select_words(r, borrow, r /* tmp < 0 */, tmp /* tmp >= 0 */, r_len);
  178|  1.36M|  return borrow;
  179|  1.36M|}
bcm.c:bn_sub_part_words:
  130|  2.72M|                                  const BN_ULONG *b, int cl, int dl) {
  131|  2.72M|  assert(cl >= 0);
  132|  2.72M|  BN_ULONG borrow = bn_sub_words(r, a, b, cl);
  133|  2.72M|  if (dl == 0) {
  ------------------
  |  Branch (133:7): [True: 2.62M, False: 95.1k]
  ------------------
  134|  2.62M|    return borrow;
  135|  2.62M|  }
  136|       |
  137|  95.1k|  r += cl;
  138|  95.1k|  a += cl;
  139|  95.1k|  b += cl;
  140|       |
  141|  95.1k|  if (dl < 0) {
  ------------------
  |  Branch (141:7): [True: 47.5k, False: 47.5k]
  ------------------
  142|       |    // |a| is shorter than |b|. Complete the subtraction as if the excess words
  143|       |    // in |a| were zeros.
  144|  47.5k|    dl = -dl;
  145|   883k|    for (int i = 0; i < dl; i++) {
  ------------------
  |  Branch (145:21): [True: 836k, False: 47.5k]
  ------------------
  146|   836k|      r[i] = 0u - b[i] - borrow;
  147|   836k|      borrow |= r[i] != 0;
  148|   836k|    }
  149|  47.5k|  } else {
  150|       |    // |b| is shorter than |a|. Complete the subtraction as if the excess words
  151|       |    // in |b| were zeros.
  152|   883k|    for (int i = 0; i < dl; i++) {
  ------------------
  |  Branch (152:21): [True: 836k, False: 47.5k]
  ------------------
  153|       |      // |r| and |a| may alias, so use a temporary.
  154|   836k|      BN_ULONG tmp = a[i];
  155|   836k|      r[i] = a[i] - borrow;
  156|   836k|      borrow = tmp < r[i];
  157|   836k|    }
  158|  47.5k|  }
  159|       |
  160|  95.1k|  return borrow;
  161|  2.72M|}
bcm.c:bn_mul_impl:
  420|   854k|                       BN_CTX *ctx) {
  421|   854k|  int al = a->width;
  422|   854k|  int bl = b->width;
  423|   854k|  if (al == 0 || bl == 0) {
  ------------------
  |  Branch (423:7): [True: 57.8k, False: 796k]
  |  Branch (423:18): [True: 5.57k, False: 790k]
  ------------------
  424|  63.4k|    BN_zero(r);
  425|  63.4k|    return 1;
  426|  63.4k|  }
  427|       |
  428|   790k|  int ret = 0;
  429|   790k|  BIGNUM *rr;
  430|   790k|  BN_CTX_start(ctx);
  431|   790k|  if (r == a || r == b) {
  ------------------
  |  Branch (431:7): [True: 563k, False: 227k]
  |  Branch (431:17): [True: 0, False: 227k]
  ------------------
  432|   563k|    rr = BN_CTX_get(ctx);
  433|   563k|    if (rr == NULL) {
  ------------------
  |  Branch (433:9): [True: 0, False: 563k]
  ------------------
  434|      0|      goto err;
  435|      0|    }
  436|   563k|  } else {
  437|   227k|    rr = r;
  438|   227k|  }
  439|   790k|  rr->neg = a->neg ^ b->neg;
  440|       |
  441|   790k|  int i = al - bl;
  442|   790k|  if (i == 0) {
  ------------------
  |  Branch (442:7): [True: 749k, False: 41.2k]
  ------------------
  443|   749k|    if (al == 8) {
  ------------------
  |  Branch (443:9): [True: 38.1k, False: 711k]
  ------------------
  444|  38.1k|      if (!bn_wexpand(rr, 16)) {
  ------------------
  |  Branch (444:11): [True: 0, False: 38.1k]
  ------------------
  445|      0|        goto err;
  446|      0|      }
  447|  38.1k|      rr->width = 16;
  448|  38.1k|      bn_mul_comba8(rr->d, a->d, b->d);
  449|  38.1k|      goto end;
  450|  38.1k|    }
  451|   749k|  }
  452|       |
  453|   752k|  int top = al + bl;
  454|   752k|  static const int kMulNormalSize = 16;
  455|   752k|  if (al >= kMulNormalSize && bl >= kMulNormalSize) {
  ------------------
  |  Branch (455:7): [True: 347k, False: 405k]
  |  Branch (455:31): [True: 340k, False: 7.17k]
  ------------------
  456|   340k|    if (-1 <= i && i <= 1) {
  ------------------
  |  Branch (456:9): [True: 340k, False: 489]
  |  Branch (456:20): [True: 339k, False: 659]
  ------------------
  457|       |      // Find the largest power of two less than or equal to the larger length.
  458|   339k|      int j;
  459|   339k|      if (i >= 0) {
  ------------------
  |  Branch (459:11): [True: 333k, False: 6.16k]
  ------------------
  460|   333k|        j = BN_num_bits_word((BN_ULONG)al);
  461|   333k|      } else {
  462|  6.16k|        j = BN_num_bits_word((BN_ULONG)bl);
  463|  6.16k|      }
  464|   339k|      j = 1 << (j - 1);
  465|   339k|      assert(j <= al || j <= bl);
  466|   339k|      BIGNUM *t = BN_CTX_get(ctx);
  467|   339k|      if (t == NULL) {
  ------------------
  |  Branch (467:11): [True: 0, False: 339k]
  ------------------
  468|      0|        goto err;
  469|      0|      }
  470|   339k|      if (al > j || bl > j) {
  ------------------
  |  Branch (470:11): [True: 15.4k, False: 323k]
  |  Branch (470:21): [True: 4.49k, False: 319k]
  ------------------
  471|       |        // We know |al| and |bl| are at most one from each other, so if al > j,
  472|       |        // bl >= j, and vice versa. Thus we can use |bn_mul_part_recursive|.
  473|       |        //
  474|       |        // TODO(davidben): This codepath is almost unused in standard
  475|       |        // algorithms. Is this optimization necessary? See notes in
  476|       |        // https://boringssl-review.googlesource.com/q/I0bd604e2cd6a75c266f64476c23a730ca1721ea6
  477|  19.9k|        assert(al >= j && bl >= j);
  478|  19.9k|        if (!bn_wexpand(t, j * 8) ||
  ------------------
  |  Branch (478:13): [True: 0, False: 19.9k]
  ------------------
  479|  19.9k|            !bn_wexpand(rr, j * 4)) {
  ------------------
  |  Branch (479:13): [True: 0, False: 19.9k]
  ------------------
  480|      0|          goto err;
  481|      0|        }
  482|  19.9k|        bn_mul_part_recursive(rr->d, a->d, b->d, j, al - j, bl - j, t->d);
  483|   319k|      } else {
  484|       |        // al <= j && bl <= j. Additionally, we know j <= al or j <= bl, so one
  485|       |        // of al - j or bl - j is zero. The other, by the bound on |i| above, is
  486|       |        // zero or -1. Thus, we can use |bn_mul_recursive|.
  487|   319k|        if (!bn_wexpand(t, j * 4) ||
  ------------------
  |  Branch (487:13): [True: 0, False: 319k]
  ------------------
  488|   319k|            !bn_wexpand(rr, j * 2)) {
  ------------------
  |  Branch (488:13): [True: 0, False: 319k]
  ------------------
  489|      0|          goto err;
  490|      0|        }
  491|   319k|        bn_mul_recursive(rr->d, a->d, b->d, j, al - j, bl - j, t->d);
  492|   319k|      }
  493|   339k|      rr->width = top;
  494|   339k|      goto end;
  495|   339k|    }
  496|   340k|  }
  497|       |
  498|   413k|  if (!bn_wexpand(rr, top)) {
  ------------------
  |  Branch (498:7): [True: 0, False: 413k]
  ------------------
  499|      0|    goto err;
  500|      0|  }
  501|   413k|  rr->width = top;
  502|   413k|  bn_mul_normal(rr->d, a->d, al, b->d, bl);
  503|       |
  504|   790k|end:
  505|   790k|  if (r != rr && !BN_copy(r, rr)) {
  ------------------
  |  Branch (505:7): [True: 563k, False: 227k]
  |  Branch (505:18): [True: 0, False: 563k]
  ------------------
  506|      0|    goto err;
  507|      0|  }
  508|   790k|  ret = 1;
  509|       |
  510|   790k|err:
  511|   790k|  BN_CTX_end(ctx);
  512|   790k|  return ret;
  513|   790k|}
bcm.c:bn_mul_part_recursive:
  312|  21.1k|                                  BN_ULONG *t) {
  313|       |  // |n| is a power of two.
  314|  21.1k|  assert(n != 0 && (n & (n - 1)) == 0);
  315|       |  // Check |tna| and |tnb| are in range.
  316|  21.1k|  assert(0 <= tna && tna < n);
  317|  21.1k|  assert(0 <= tnb && tnb < n);
  318|  21.1k|  assert(-1 <= tna - tnb && tna - tnb <= 1);
  319|       |
  320|  21.1k|  int n2 = n * 2;
  321|  21.1k|  if (n < 8) {
  ------------------
  |  Branch (321:7): [True: 0, False: 21.1k]
  ------------------
  322|      0|    bn_mul_normal(r, a, n + tna, b, n + tnb);
  323|      0|    OPENSSL_memset(r + n2 + tna + tnb, 0, n2 - tna - tnb);
  324|      0|    return;
  325|      0|  }
  326|       |
  327|       |  // Split |a| and |b| into a0,a1 and b0,b1, where a0 and b0 have size |n|. |a1|
  328|       |  // and |b1| have size |tna| and |tnb|, respectively.
  329|       |  // Split |t| into t0,t1,t2,t3, each of size |n|, with the remaining 4*|n| used
  330|       |  // for recursive calls.
  331|       |  // Split |r| into r0,r1,r2,r3. We must contribute a0*b0 to r0,r1, a0*a1+b0*b1
  332|       |  // to r1,r2, and a1*b1 to r2,r3. The middle term we will compute as:
  333|       |  //
  334|       |  //   a0*a1 + b0*b1 = (a0 - a1)*(b1 - b0) + a1*b1 + a0*b0
  335|       |
  336|       |  // t0 = a0 - a1 and t1 = b1 - b0. The result will be multiplied, so we XOR
  337|       |  // their sign masks, giving the sign of (a0 - a1)*(b1 - b0). t0 and t1
  338|       |  // themselves store the absolute value.
  339|  21.1k|  BN_ULONG neg = bn_abs_sub_part_words(t, a, &a[n], tna, n - tna, &t[n2]);
  340|  21.1k|  neg ^= bn_abs_sub_part_words(&t[n], &b[n], b, tnb, tnb - n, &t[n2]);
  341|       |
  342|       |  // Compute:
  343|       |  // t2,t3 = t0 * t1 = |(a0 - a1)*(b1 - b0)|
  344|       |  // r0,r1 = a0 * b0
  345|       |  // r2,r3 = a1 * b1
  346|  21.1k|  if (n == 8) {
  ------------------
  |  Branch (346:7): [True: 0, False: 21.1k]
  ------------------
  347|      0|    bn_mul_comba8(&t[n2], t, &t[n]);
  348|      0|    bn_mul_comba8(r, a, b);
  349|       |
  350|      0|    bn_mul_normal(&r[n2], &a[n], tna, &b[n], tnb);
  351|       |    // |bn_mul_normal| only writes |tna| + |tna| words. Zero the rest.
  352|      0|    OPENSSL_memset(&r[n2 + tna + tnb], 0, sizeof(BN_ULONG) * (n2 - tna - tnb));
  353|  21.1k|  } else {
  354|  21.1k|    BN_ULONG *p = &t[n2 * 2];
  355|  21.1k|    bn_mul_recursive(&t[n2], t, &t[n], n, 0, 0, p);
  356|  21.1k|    bn_mul_recursive(r, a, b, n, 0, 0, p);
  357|       |
  358|  21.1k|    OPENSSL_memset(&r[n2], 0, sizeof(BN_ULONG) * n2);
  359|  21.1k|    if (tna < BN_MUL_RECURSIVE_SIZE_NORMAL &&
  ------------------
  |  |   70|  42.2k|#define BN_MUL_RECURSIVE_SIZE_NORMAL 16
  ------------------
  |  Branch (359:9): [True: 17.7k, False: 3.33k]
  ------------------
  360|  21.1k|        tnb < BN_MUL_RECURSIVE_SIZE_NORMAL) {
  ------------------
  |  |   70|  17.7k|#define BN_MUL_RECURSIVE_SIZE_NORMAL 16
  ------------------
  |  Branch (360:9): [True: 17.5k, False: 246]
  ------------------
  361|  17.5k|      bn_mul_normal(&r[n2], &a[n], tna, &b[n], tnb);
  362|  17.5k|    } else {
  363|  3.57k|      int i = n;
  364|  3.57k|      for (;;) {
  365|  3.57k|        i /= 2;
  366|  3.57k|        if (i < tna || i < tnb) {
  ------------------
  |  Branch (366:13): [True: 969, False: 2.61k]
  |  Branch (366:24): [True: 208, False: 2.40k]
  ------------------
  367|       |          // E.g., n == 16, i == 8 and tna == 11. |tna| and |tnb| are within one
  368|       |          // of each other, so if |tna| is larger and tna > i, then we know
  369|       |          // tnb >= i, and this call is valid.
  370|  1.17k|          bn_mul_part_recursive(&r[n2], &a[n], &b[n], i, tna - i, tnb - i, p);
  371|  1.17k|          break;
  372|  1.17k|        }
  373|  2.40k|        if (i == tna || i == tnb) {
  ------------------
  |  Branch (373:13): [True: 2.15k, False: 246]
  |  Branch (373:25): [True: 246, False: 0]
  ------------------
  374|       |          // If there is only a bottom half to the number, just do it. We know
  375|       |          // the larger of |tna - i| and |tnb - i| is zero. The other is zero or
  376|       |          // -1 by because of |tna| and |tnb| differ by at most one.
  377|  2.40k|          bn_mul_recursive(&r[n2], &a[n], &b[n], i, tna - i, tnb - i, p);
  378|  2.40k|          break;
  379|  2.40k|        }
  380|       |
  381|       |        // This loop will eventually terminate when |i| falls below
  382|       |        // |BN_MUL_RECURSIVE_SIZE_NORMAL| because we know one of |tna| and |tnb|
  383|       |        // exceeds that.
  384|  2.40k|      }
  385|  3.57k|    }
  386|  21.1k|  }
  387|       |
  388|       |  // t0,t1,c = r0,r1 + r2,r3 = a0*b0 + a1*b1
  389|  21.1k|  BN_ULONG c = bn_add_words(t, r, &r[n2], n2);
  390|       |
  391|       |  // t2,t3,c = t0,t1,c + neg*t2,t3 = (a0 - a1)*(b1 - b0) + a1*b1 + a0*b0.
  392|       |  // The second term is stored as the absolute value, so we do this with a
  393|       |  // constant-time select.
  394|  21.1k|  BN_ULONG c_neg = c - bn_sub_words(&t[n2 * 2], t, &t[n2], n2);
  395|  21.1k|  BN_ULONG c_pos = c + bn_add_words(&t[n2], t, &t[n2], n2);
  396|  21.1k|  bn_select_words(&t[n2], neg, &t[n2 * 2], &t[n2], n2);
  397|  21.1k|  static_assert(sizeof(BN_ULONG) <= sizeof(crypto_word_t),
  398|  21.1k|                "crypto_word_t is too small");
  399|  21.1k|  c = constant_time_select_w(neg, c_neg, c_pos);
  400|       |
  401|       |  // We now have our three components. Add them together.
  402|       |  // r1,r2,c = r1,r2 + t2,t3,c
  403|  21.1k|  c += bn_add_words(&r[n], &r[n], &t[n2], n2);
  404|       |
  405|       |  // Propagate the carry bit to the end.
  406|   549k|  for (int i = n + n2; i < n2 + n2; i++) {
  ------------------
  |  Branch (406:24): [True: 528k, False: 21.1k]
  ------------------
  407|   528k|    BN_ULONG old = r[i];
  408|   528k|    r[i] = old + c;
  409|   528k|    c = r[i] < old;
  410|   528k|  }
  411|       |
  412|       |  // The product should fit without carries.
  413|  21.1k|  assert(c == 0);
  414|  21.1k|}
bcm.c:bn_mul_recursive:
  211|   666k|                             int n2, int dna, int dnb, BN_ULONG *t) {
  212|       |  // |n2| is a power of two.
  213|   666k|  assert(n2 != 0 && (n2 & (n2 - 1)) == 0);
  214|       |  // Check |dna| and |dnb| are in range.
  215|   666k|  assert(-BN_MUL_RECURSIVE_SIZE_NORMAL/2 <= dna && dna <= 0);
  216|   666k|  assert(-BN_MUL_RECURSIVE_SIZE_NORMAL/2 <= dnb && dnb <= 0);
  217|       |
  218|       |  // Only call bn_mul_comba 8 if n2 == 8 and the
  219|       |  // two arrays are complete [steve]
  220|   666k|  if (n2 == 8 && dna == 0 && dnb == 0) {
  ------------------
  |  Branch (220:7): [True: 6.84k, False: 659k]
  |  Branch (220:18): [True: 5.61k, False: 1.22k]
  |  Branch (220:30): [True: 4.56k, False: 1.05k]
  ------------------
  221|  4.56k|    bn_mul_comba8(r, a, b);
  222|  4.56k|    return;
  223|  4.56k|  }
  224|       |
  225|       |  // Else do normal multiply
  226|   661k|  if (n2 < BN_MUL_RECURSIVE_SIZE_NORMAL) {
  ------------------
  |  |   70|   661k|#define BN_MUL_RECURSIVE_SIZE_NORMAL 16
  ------------------
  |  Branch (226:7): [True: 2.28k, False: 659k]
  ------------------
  227|  2.28k|    bn_mul_normal(r, a, n2 + dna, b, n2 + dnb);
  228|  2.28k|    if (dna + dnb < 0) {
  ------------------
  |  Branch (228:9): [True: 2.28k, False: 0]
  ------------------
  229|  2.28k|      OPENSSL_memset(&r[2 * n2 + dna + dnb], 0,
  230|  2.28k|                     sizeof(BN_ULONG) * -(dna + dnb));
  231|  2.28k|    }
  232|  2.28k|    return;
  233|  2.28k|  }
  234|       |
  235|       |  // Split |a| and |b| into a0,a1 and b0,b1, where a0 and b0 have size |n|.
  236|       |  // Split |t| into t0,t1,t2,t3, each of size |n|, with the remaining 4*|n| used
  237|       |  // for recursive calls.
  238|       |  // Split |r| into r0,r1,r2,r3. We must contribute a0*b0 to r0,r1, a0*a1+b0*b1
  239|       |  // to r1,r2, and a1*b1 to r2,r3. The middle term we will compute as:
  240|       |  //
  241|       |  //   a0*a1 + b0*b1 = (a0 - a1)*(b1 - b0) + a1*b1 + a0*b0
  242|       |  //
  243|       |  // Note that we know |n| >= |BN_MUL_RECURSIVE_SIZE_NORMAL|/2 above, so
  244|       |  // |tna| and |tnb| are non-negative.
  245|   659k|  int n = n2 / 2, tna = n + dna, tnb = n + dnb;
  246|       |
  247|       |  // t0 = a0 - a1 and t1 = b1 - b0. The result will be multiplied, so we XOR
  248|       |  // their sign masks, giving the sign of (a0 - a1)*(b1 - b0). t0 and t1
  249|       |  // themselves store the absolute value.
  250|   659k|  BN_ULONG neg = bn_abs_sub_part_words(t, a, &a[n], tna, n - tna, &t[n2]);
  251|   659k|  neg ^= bn_abs_sub_part_words(&t[n], &b[n], b, tnb, tnb - n, &t[n2]);
  252|       |
  253|       |  // Compute:
  254|       |  // t2,t3 = t0 * t1 = |(a0 - a1)*(b1 - b0)|
  255|       |  // r0,r1 = a0 * b0
  256|       |  // r2,r3 = a1 * b1
  257|   659k|  if (n == 4 && dna == 0 && dnb == 0) {
  ------------------
  |  Branch (257:7): [True: 0, False: 659k]
  |  Branch (257:17): [True: 0, False: 0]
  |  Branch (257:29): [True: 0, False: 0]
  ------------------
  258|      0|    bn_mul_comba4(&t[n2], t, &t[n]);
  259|       |
  260|      0|    bn_mul_comba4(r, a, b);
  261|      0|    bn_mul_comba4(&r[n2], &a[n], &b[n]);
  262|   659k|  } else if (n == 8 && dna == 0 && dnb == 0) {
  ------------------
  |  Branch (262:14): [True: 561k, False: 98.4k]
  |  Branch (262:24): [True: 559k, False: 1.22k]
  |  Branch (262:36): [True: 558k, False: 1.05k]
  ------------------
  263|   558k|    bn_mul_comba8(&t[n2], t, &t[n]);
  264|       |
  265|   558k|    bn_mul_comba8(r, a, b);
  266|   558k|    bn_mul_comba8(&r[n2], &a[n], &b[n]);
  267|   558k|  } else {
  268|   100k|    BN_ULONG *p = &t[n2 * 2];
  269|   100k|    bn_mul_recursive(&t[n2], t, &t[n], n, 0, 0, p);
  270|   100k|    bn_mul_recursive(r, a, b, n, 0, 0, p);
  271|   100k|    bn_mul_recursive(&r[n2], &a[n], &b[n], n, dna, dnb, p);
  272|   100k|  }
  273|       |
  274|       |  // t0,t1,c = r0,r1 + r2,r3 = a0*b0 + a1*b1
  275|   659k|  BN_ULONG c = bn_add_words(t, r, &r[n2], n2);
  276|       |
  277|       |  // t2,t3,c = t0,t1,c + neg*t2,t3 = (a0 - a1)*(b1 - b0) + a1*b1 + a0*b0.
  278|       |  // The second term is stored as the absolute value, so we do this with a
  279|       |  // constant-time select.
  280|   659k|  BN_ULONG c_neg = c - bn_sub_words(&t[n2 * 2], t, &t[n2], n2);
  281|   659k|  BN_ULONG c_pos = c + bn_add_words(&t[n2], t, &t[n2], n2);
  282|   659k|  bn_select_words(&t[n2], neg, &t[n2 * 2], &t[n2], n2);
  283|   659k|  static_assert(sizeof(BN_ULONG) <= sizeof(crypto_word_t),
  284|   659k|                "crypto_word_t is too small");
  285|   659k|  c = constant_time_select_w(neg, c_neg, c_pos);
  286|       |
  287|       |  // We now have our three components. Add them together.
  288|       |  // r1,r2,c = r1,r2 + t2,t3,c
  289|   659k|  c += bn_add_words(&r[n], &r[n], &t[n2], n2);
  290|       |
  291|       |  // Propagate the carry bit to the end.
  292|  6.85M|  for (int i = n + n2; i < n2 + n2; i++) {
  ------------------
  |  Branch (292:24): [True: 6.19M, False: 659k]
  ------------------
  293|  6.19M|    BN_ULONG old = r[i];
  294|  6.19M|    r[i] = old + c;
  295|  6.19M|    c = r[i] < old;
  296|  6.19M|  }
  297|       |
  298|       |  // The product should fit without carries.
  299|   659k|  assert(c == 0);
  300|   659k|}
bcm.c:bn_mul_normal:
   82|   433k|                          const BN_ULONG *b, size_t nb) {
   83|   433k|  if (na < nb) {
  ------------------
  |  Branch (83:7): [True: 21.8k, False: 411k]
  ------------------
   84|  21.8k|    size_t itmp = na;
   85|  21.8k|    na = nb;
   86|  21.8k|    nb = itmp;
   87|  21.8k|    const BN_ULONG *ltmp = a;
   88|  21.8k|    a = b;
   89|  21.8k|    b = ltmp;
   90|  21.8k|  }
   91|   433k|  BN_ULONG *rr = &(r[na]);
   92|   433k|  if (nb == 0) {
  ------------------
  |  Branch (92:7): [True: 9.03k, False: 424k]
  ------------------
   93|  9.03k|    OPENSSL_memset(r, 0, na * sizeof(BN_ULONG));
   94|  9.03k|    return;
   95|  9.03k|  }
   96|   424k|  rr[0] = bn_mul_words(r, a, na, b[0]);
   97|       |
   98|   485k|  for (;;) {
   99|   485k|    if (--nb == 0) {
  ------------------
  |  Branch (99:9): [True: 364k, False: 120k]
  ------------------
  100|   364k|      return;
  101|   364k|    }
  102|   120k|    rr[1] = bn_mul_add_words(&(r[1]), a, na, b[1]);
  103|   120k|    if (--nb == 0) {
  ------------------
  |  Branch (103:9): [True: 35.4k, False: 85.2k]
  ------------------
  104|  35.4k|      return;
  105|  35.4k|    }
  106|  85.2k|    rr[2] = bn_mul_add_words(&(r[2]), a, na, b[2]);
  107|  85.2k|    if (--nb == 0) {
  ------------------
  |  Branch (107:9): [True: 18.8k, False: 66.4k]
  ------------------
  108|  18.8k|      return;
  109|  18.8k|    }
  110|  66.4k|    rr[3] = bn_mul_add_words(&(r[3]), a, na, b[3]);
  111|  66.4k|    if (--nb == 0) {
  ------------------
  |  Branch (111:9): [True: 5.31k, False: 61.0k]
  ------------------
  112|  5.31k|      return;
  113|  5.31k|    }
  114|  61.0k|    rr[4] = bn_mul_add_words(&(r[4]), a, na, b[4]);
  115|  61.0k|    rr += 4;
  116|  61.0k|    r += 4;
  117|  61.0k|    b += 4;
  118|  61.0k|  }
  119|   424k|}
bcm.c:bn_sqr_normal:
  551|   358k|                          BN_ULONG *tmp) {
  552|   358k|  if (n == 0) {
  ------------------
  |  Branch (552:7): [True: 0, False: 358k]
  ------------------
  553|      0|    return;
  554|      0|  }
  555|       |
  556|   358k|  size_t max = n * 2;
  557|   358k|  const BN_ULONG *ap = a;
  558|   358k|  BN_ULONG *rp = r;
  559|   358k|  rp[0] = rp[max - 1] = 0;
  560|   358k|  rp++;
  561|       |
  562|       |  // Compute the contribution of a[i] * a[j] for all i < j.
  563|   358k|  if (n > 1) {
  ------------------
  |  Branch (563:7): [True: 9.08k, False: 349k]
  ------------------
  564|  9.08k|    ap++;
  565|  9.08k|    rp[n - 1] = bn_mul_words(rp, ap, n - 1, ap[-1]);
  566|  9.08k|    rp += 2;
  567|  9.08k|  }
  568|   358k|  if (n > 2) {
  ------------------
  |  Branch (568:7): [True: 6.47k, False: 352k]
  ------------------
  569|   102k|    for (size_t i = n - 2; i > 0; i--) {
  ------------------
  |  Branch (569:28): [True: 95.5k, False: 6.47k]
  ------------------
  570|  95.5k|      ap++;
  571|  95.5k|      rp[i] = bn_mul_add_words(rp, ap, i, ap[-1]);
  572|  95.5k|      rp += 2;
  573|  95.5k|    }
  574|  6.47k|  }
  575|       |
  576|       |  // The final result fits in |max| words, so none of the following operations
  577|       |  // will overflow.
  578|       |
  579|       |  // Double |r|, giving the contribution of a[i] * a[j] for all i != j.
  580|   358k|  bn_add_words(r, r, r, max);
  581|       |
  582|       |  // Add in the contribution of a[i] * a[i] for all i.
  583|   358k|  bn_sqr_words(tmp, a, n);
  584|   358k|  bn_add_words(r, r, tmp, max);
  585|   358k|}
bcm.c:bn_sqr_recursive:
  591|   247k|                             BN_ULONG *t) {
  592|       |  // |n2| is a power of two.
  593|   247k|  assert(n2 != 0 && (n2 & (n2 - 1)) == 0);
  594|       |
  595|   247k|  if (n2 == 4) {
  ------------------
  |  Branch (595:7): [True: 0, False: 247k]
  ------------------
  596|      0|    bn_sqr_comba4(r, a);
  597|      0|    return;
  598|      0|  }
  599|   247k|  if (n2 == 8) {
  ------------------
  |  Branch (599:7): [True: 170k, False: 77.0k]
  ------------------
  600|   170k|    bn_sqr_comba8(r, a);
  601|   170k|    return;
  602|   170k|  }
  603|  77.0k|  if (n2 < BN_SQR_RECURSIVE_SIZE_NORMAL) {
  ------------------
  |  |   71|  77.0k|#define BN_SQR_RECURSIVE_SIZE_NORMAL BN_MUL_RECURSIVE_SIZE_NORMAL
  |  |  ------------------
  |  |  |  |   70|  77.0k|#define BN_MUL_RECURSIVE_SIZE_NORMAL 16
  |  |  ------------------
  ------------------
  |  Branch (603:7): [True: 0, False: 77.0k]
  ------------------
  604|      0|    bn_sqr_normal(r, a, n2, t);
  605|      0|    return;
  606|      0|  }
  607|       |
  608|       |  // Split |a| into a0,a1, each of size |n|.
  609|       |  // Split |t| into t0,t1,t2,t3, each of size |n|, with the remaining 4*|n| used
  610|       |  // for recursive calls.
  611|       |  // Split |r| into r0,r1,r2,r3. We must contribute a0^2 to r0,r1, 2*a0*a1 to
  612|       |  // r1,r2, and a1^2 to r2,r3.
  613|  77.0k|  size_t n = n2 / 2;
  614|  77.0k|  BN_ULONG *t_recursive = &t[n2 * 2];
  615|       |
  616|       |  // t0 = |a0 - a1|.
  617|  77.0k|  bn_abs_sub_words(t, a, &a[n], n, &t[n]);
  618|       |  // t2,t3 = t0^2 = |a0 - a1|^2 = a0^2 - 2*a0*a1 + a1^2
  619|  77.0k|  bn_sqr_recursive(&t[n2], t, n, t_recursive);
  620|       |
  621|       |  // r0,r1 = a0^2
  622|  77.0k|  bn_sqr_recursive(r, a, n, t_recursive);
  623|       |
  624|       |  // r2,r3 = a1^2
  625|  77.0k|  bn_sqr_recursive(&r[n2], &a[n], n, t_recursive);
  626|       |
  627|       |  // t0,t1,c = r0,r1 + r2,r3 = a0^2 + a1^2
  628|  77.0k|  BN_ULONG c = bn_add_words(t, r, &r[n2], n2);
  629|       |  // t2,t3,c = t0,t1,c - t2,t3 = 2*a0*a1
  630|  77.0k|  c -= bn_sub_words(&t[n2], t, &t[n2], n2);
  631|       |
  632|       |  // We now have our three components. Add them together.
  633|       |  // r1,r2,c = r1,r2 + t2,t3,c
  634|  77.0k|  c += bn_add_words(&r[n], &r[n], &t[n2], n2);
  635|       |
  636|       |  // Propagate the carry bit to the end.
  637|   889k|  for (size_t i = n + n2; i < n2 + n2; i++) {
  ------------------
  |  Branch (637:27): [True: 812k, False: 77.0k]
  ------------------
  638|   812k|    BN_ULONG old = r[i];
  639|   812k|    r[i] = old + c;
  640|   812k|    c = r[i] < old;
  641|   812k|  }
  642|       |
  643|       |  // The square should fit without carries.
  644|  77.0k|  assert(c == 0);
  645|  77.0k|}
bcm.c:bn_abs_sub_words:
   75|  77.0k|                             size_t num, BN_ULONG *tmp) {
   76|  77.0k|  BN_ULONG borrow = bn_sub_words(tmp, a, b, num);
   77|  77.0k|  bn_sub_words(r, b, a, num);
   78|  77.0k|  bn_select_words(r, 0 - borrow, r /* tmp < 0 */, tmp /* tmp >= 0 */, num);
   79|  77.0k|}

bcm.c:rsaz_avx2_preferred:
   46|     69|OPENSSL_INLINE int rsaz_avx2_preferred(void) {
   47|     69|  if (CRYPTO_is_BMI1_capable() && CRYPTO_is_BMI2_capable() &&
  ------------------
  |  Branch (47:7): [True: 69, False: 0]
  |  Branch (47:35): [True: 69, False: 0]
  ------------------
   48|     69|      CRYPTO_is_ADX_capable()) {
  ------------------
  |  Branch (48:7): [True: 69, False: 0]
  ------------------
   49|       |    // If BMI1, BMI2, and ADX are available, x86_64-mont5.pl is faster. See the
   50|       |    // .Lmulx4x_enter and .Lpowerx5_enter branches.
   51|     69|    return 0;
   52|     69|  }
   53|      0|  return CRYPTO_is_AVX2_capable();
   54|     69|}

BN_lshift:
   67|  1.24M|int BN_lshift(BIGNUM *r, const BIGNUM *a, int n) {
   68|  1.24M|  int i, nw, lb, rb;
   69|  1.24M|  BN_ULONG *t, *f;
   70|  1.24M|  BN_ULONG l;
   71|       |
   72|  1.24M|  if (n < 0) {
  ------------------
  |  Branch (72:7): [True: 0, False: 1.24M]
  ------------------
   73|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
   74|      0|    return 0;
   75|      0|  }
   76|       |
   77|  1.24M|  r->neg = a->neg;
   78|  1.24M|  nw = n / BN_BITS2;
  ------------------
  |  |  151|  1.24M|#define BN_BITS2 64
  ------------------
   79|  1.24M|  if (!bn_wexpand(r, a->width + nw + 1)) {
  ------------------
  |  Branch (79:7): [True: 0, False: 1.24M]
  ------------------
   80|      0|    return 0;
   81|      0|  }
   82|  1.24M|  lb = n % BN_BITS2;
  ------------------
  |  |  151|  1.24M|#define BN_BITS2 64
  ------------------
   83|  1.24M|  rb = BN_BITS2 - lb;
  ------------------
  |  |  151|  1.24M|#define BN_BITS2 64
  ------------------
   84|  1.24M|  f = a->d;
   85|  1.24M|  t = r->d;
   86|  1.24M|  t[a->width + nw] = 0;
   87|  1.24M|  if (lb == 0) {
  ------------------
  |  Branch (87:7): [True: 204k, False: 1.04M]
  ------------------
   88|  4.69M|    for (i = a->width - 1; i >= 0; i--) {
  ------------------
  |  Branch (88:28): [True: 4.48M, False: 204k]
  ------------------
   89|  4.48M|      t[nw + i] = f[i];
   90|  4.48M|    }
   91|  1.04M|  } else {
   92|  14.6M|    for (i = a->width - 1; i >= 0; i--) {
  ------------------
  |  Branch (92:28): [True: 13.6M, False: 1.04M]
  ------------------
   93|  13.6M|      l = f[i];
   94|  13.6M|      t[nw + i + 1] |= l >> rb;
   95|  13.6M|      t[nw + i] = l << lb;
   96|  13.6M|    }
   97|  1.04M|  }
   98|  1.24M|  OPENSSL_memset(t, 0, nw * sizeof(t[0]));
   99|  1.24M|  r->width = a->width + nw + 1;
  100|  1.24M|  bn_set_minimal_width(r);
  101|       |
  102|  1.24M|  return 1;
  103|  1.24M|}
bn_rshift_words:
  137|   779k|                     size_t num) {
  138|   779k|  unsigned shift_bits = shift % BN_BITS2;
  ------------------
  |  |  151|   779k|#define BN_BITS2 64
  ------------------
  139|   779k|  size_t shift_words = shift / BN_BITS2;
  ------------------
  |  |  151|   779k|#define BN_BITS2 64
  ------------------
  140|   779k|  if (shift_words >= num) {
  ------------------
  |  Branch (140:7): [True: 56.8k, False: 723k]
  ------------------
  141|  56.8k|    OPENSSL_memset(r, 0, num * sizeof(BN_ULONG));
  142|  56.8k|    return;
  143|  56.8k|  }
  144|   723k|  if (shift_bits == 0) {
  ------------------
  |  Branch (144:7): [True: 109k, False: 613k]
  ------------------
  145|   109k|    OPENSSL_memmove(r, a + shift_words, (num - shift_words) * sizeof(BN_ULONG));
  146|   613k|  } else {
  147|  6.21M|    for (size_t i = shift_words; i < num - 1; i++) {
  ------------------
  |  Branch (147:34): [True: 5.60M, False: 613k]
  ------------------
  148|  5.60M|      r[i - shift_words] =
  149|  5.60M|          (a[i] >> shift_bits) | (a[i + 1] << (BN_BITS2 - shift_bits));
  ------------------
  |  |  151|  5.60M|#define BN_BITS2 64
  ------------------
  150|  5.60M|    }
  151|   613k|    r[num - 1 - shift_words] = a[num - 1] >> shift_bits;
  152|   613k|  }
  153|   723k|  OPENSSL_memset(r + num - shift_words, 0, shift_words * sizeof(BN_ULONG));
  154|   723k|}
BN_rshift:
  156|   779k|int BN_rshift(BIGNUM *r, const BIGNUM *a, int n) {
  157|   779k|  if (n < 0) {
  ------------------
  |  Branch (157:7): [True: 0, False: 779k]
  ------------------
  158|      0|    OPENSSL_PUT_ERROR(BN, BN_R_NEGATIVE_NUMBER);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  159|      0|    return 0;
  160|      0|  }
  161|       |
  162|   779k|  if (!bn_wexpand(r, a->width)) {
  ------------------
  |  Branch (162:7): [True: 0, False: 779k]
  ------------------
  163|      0|    return 0;
  164|      0|  }
  165|   779k|  bn_rshift_words(r->d, a->d, n, a->width);
  166|   779k|  r->neg = a->neg;
  167|   779k|  r->width = a->width;
  168|   779k|  bn_set_minimal_width(r);
  169|   779k|  return 1;
  170|   779k|}
bn_rshift1_words:
  200|   438k|void bn_rshift1_words(BN_ULONG *r, const BN_ULONG *a, size_t num) {
  201|   438k|  if (num == 0) {
  ------------------
  |  Branch (201:7): [True: 0, False: 438k]
  ------------------
  202|      0|    return;
  203|      0|  }
  204|  5.26M|  for (size_t i = 0; i < num - 1; i++) {
  ------------------
  |  Branch (204:22): [True: 4.82M, False: 438k]
  ------------------
  205|  4.82M|    r[i] = (a[i] >> 1) | (a[i + 1] << (BN_BITS2 - 1));
  ------------------
  |  |  151|  4.82M|#define BN_BITS2 64
  ------------------
  206|  4.82M|  }
  207|   438k|  r[num - 1] = a[num - 1] >> 1;
  208|   438k|}
BN_rshift1:
  210|   438k|int BN_rshift1(BIGNUM *r, const BIGNUM *a) {
  211|   438k|  if (!bn_wexpand(r, a->width)) {
  ------------------
  |  Branch (211:7): [True: 0, False: 438k]
  ------------------
  212|      0|    return 0;
  213|      0|  }
  214|   438k|  bn_rshift1_words(r->d, a->d, a->width);
  215|   438k|  r->width = a->width;
  216|   438k|  r->neg = a->neg;
  217|   438k|  bn_set_minimal_width(r);
  218|   438k|  return 1;
  219|   438k|}
BN_set_bit:
  221|  3.69k|int BN_set_bit(BIGNUM *a, int n) {
  222|  3.69k|  if (n < 0) {
  ------------------
  |  Branch (222:7): [True: 0, False: 3.69k]
  ------------------
  223|      0|    return 0;
  224|      0|  }
  225|       |
  226|  3.69k|  int i = n / BN_BITS2;
  ------------------
  |  |  151|  3.69k|#define BN_BITS2 64
  ------------------
  227|  3.69k|  int j = n % BN_BITS2;
  ------------------
  |  |  151|  3.69k|#define BN_BITS2 64
  ------------------
  228|  3.69k|  if (a->width <= i) {
  ------------------
  |  Branch (228:7): [True: 3.69k, False: 0]
  ------------------
  229|  3.69k|    if (!bn_wexpand(a, i + 1)) {
  ------------------
  |  Branch (229:9): [True: 0, False: 3.69k]
  ------------------
  230|      0|      return 0;
  231|      0|    }
  232|  45.8k|    for (int k = a->width; k < i + 1; k++) {
  ------------------
  |  Branch (232:28): [True: 42.1k, False: 3.69k]
  ------------------
  233|  42.1k|      a->d[k] = 0;
  234|  42.1k|    }
  235|  3.69k|    a->width = i + 1;
  236|  3.69k|  }
  237|       |
  238|  3.69k|  a->d[i] |= (((BN_ULONG)1) << j);
  239|       |
  240|  3.69k|  return 1;
  241|  3.69k|}
bn_is_bit_set_words:
  261|  1.01M|int bn_is_bit_set_words(const BN_ULONG *a, size_t num, size_t bit) {
  262|  1.01M|  size_t i = bit / BN_BITS2;
  ------------------
  |  |  151|  1.01M|#define BN_BITS2 64
  ------------------
  263|  1.01M|  size_t j = bit % BN_BITS2;
  ------------------
  |  |  151|  1.01M|#define BN_BITS2 64
  ------------------
  264|  1.01M|  if (i >= num) {
  ------------------
  |  Branch (264:7): [True: 0, False: 1.01M]
  ------------------
  265|      0|    return 0;
  266|      0|  }
  267|  1.01M|  return (a[i] >> j) & 1;
  268|  1.01M|}
BN_is_bit_set:
  270|  1.01M|int BN_is_bit_set(const BIGNUM *a, int n) {
  271|  1.01M|  if (n < 0) {
  ------------------
  |  Branch (271:7): [True: 0, False: 1.01M]
  ------------------
  272|      0|    return 0;
  273|      0|  }
  274|  1.01M|  return bn_is_bit_set_words(a->d, a->width, n);
  275|  1.01M|}

bcm.c:OPENSSL_ia32cap_get:
 1270|    207|OPENSSL_INLINE const uint32_t *OPENSSL_ia32cap_get(void) {
 1271|    207|  return OPENSSL_ia32cap_P;
 1272|    207|}
bcm.c:OPENSSL_memmove:
 1047|   109k|static inline void *OPENSSL_memmove(void *dst, const void *src, size_t n) {
 1048|   109k|  if (n == 0) {
  ------------------
  |  Branch (1048:7): [True: 0, False: 109k]
  ------------------
 1049|      0|    return dst;
 1050|      0|  }
 1051|       |
 1052|   109k|  return memmove(dst, src, n);
 1053|   109k|}
bcm.c:OPENSSL_memcpy:
 1039|   779k|static inline void *OPENSSL_memcpy(void *dst, const void *src, size_t n) {
 1040|   779k|  if (n == 0) {
  ------------------
  |  Branch (1040:7): [True: 80.6k, False: 699k]
  ------------------
 1041|  80.6k|    return dst;
 1042|  80.6k|  }
 1043|       |
 1044|   699k|  return memcpy(dst, src, n);
 1045|   779k|}
bcm.c:constant_time_eq_int:
  462|   663k|static inline crypto_word_t constant_time_eq_int(int a, int b) {
  463|   663k|  return constant_time_eq_w((crypto_word_t)(a), (crypto_word_t)(b));
  464|   663k|}
bcm.c:constant_time_is_zero_w:
  427|  3.13M|static inline crypto_word_t constant_time_is_zero_w(crypto_word_t a) {
  428|       |  // Here is an SMT-LIB verification of this formula:
  429|       |  //
  430|       |  // (define-fun is_zero ((a (_ BitVec 32))) (_ BitVec 32)
  431|       |  //   (bvand (bvnot a) (bvsub a #x00000001))
  432|       |  // )
  433|       |  //
  434|       |  // (declare-fun a () (_ BitVec 32))
  435|       |  //
  436|       |  // (assert (not (= (= #x00000001 (bvlshr (is_zero a) #x0000001f)) (= a #x00000000))))
  437|       |  // (check-sat)
  438|       |  // (get-model)
  439|  3.13M|  return constant_time_msb_w(~a & (a - 1));
  440|  3.13M|}
bcm.c:constant_time_msb_w:
  368|  5.53M|static inline crypto_word_t constant_time_msb_w(crypto_word_t a) {
  369|  5.53M|  return 0u - (a >> (sizeof(a) * 8 - 1));
  370|  5.53M|}
bcm.c:constant_time_eq_w:
  450|  3.06M|                                               crypto_word_t b) {
  451|  3.06M|  return constant_time_is_zero_w(a ^ b);
  452|  3.06M|}
bcm.c:constant_time_select_w:
  477|  43.7M|                                                   crypto_word_t b) {
  478|       |  // Clang recognizes this pattern as a select. While it usually transforms it
  479|       |  // to a cmov, it sometimes further transforms it into a branch, which we do
  480|       |  // not want.
  481|       |  //
  482|       |  // Hiding the value of the mask from the compiler evades this transformation.
  483|  43.7M|  mask = value_barrier_w(mask);
  484|  43.7M|  return (mask & a) | (~mask & b);
  485|  43.7M|}
bcm.c:value_barrier_w:
  340|  44.3M|static inline crypto_word_t value_barrier_w(crypto_word_t a) {
  341|  44.3M|#if defined(__GNUC__) || defined(__clang__)
  342|  44.3M|  __asm__("" : "+r"(a) : /* no inputs */);
  343|  44.3M|#endif
  344|  44.3M|  return a;
  345|  44.3M|}
bcm.c:OPENSSL_memset:
 1055|  2.54M|static inline void *OPENSSL_memset(void *dst, int c, size_t n) {
 1056|  2.54M|  if (n == 0) {
  ------------------
  |  Branch (1056:7): [True: 1.02M, False: 1.51M]
  ------------------
 1057|  1.02M|    return dst;
 1058|  1.02M|  }
 1059|       |
 1060|  1.51M|  return memset(dst, c, n);
 1061|  2.54M|}
bcm.c:CRYPTO_load_word_be:
 1122|  23.1k|static inline crypto_word_t CRYPTO_load_word_be(const void *in) {
 1123|  23.1k|  crypto_word_t v;
 1124|  23.1k|  OPENSSL_memcpy(&v, in, sizeof(v));
 1125|  23.1k|#if defined(OPENSSL_64_BIT)
 1126|  23.1k|  static_assert(sizeof(v) == 8, "crypto_word_t has unexpected size");
 1127|  23.1k|  return CRYPTO_bswap8(v);
 1128|       |#else
 1129|       |  static_assert(sizeof(v) == 4, "crypto_word_t has unexpected size");
 1130|       |  return CRYPTO_bswap4(v);
 1131|       |#endif
 1132|  23.1k|}
bcm.c:CRYPTO_bswap8:
  949|  23.1k|static inline uint64_t CRYPTO_bswap8(uint64_t x) {
  950|  23.1k|  return __builtin_bswap64(x);
  951|  23.1k|}
bcm.c:constant_time_select_int:
  503|  4.86M|static inline int constant_time_select_int(crypto_word_t mask, int a, int b) {
  504|  4.86M|  return (int)(constant_time_select_w(mask, (crypto_word_t)(a),
  505|  4.86M|                                      (crypto_word_t)(b)));
  506|  4.86M|}
bcm.c:constant_time_declassify_int:
  572|  2.55k|static inline int constant_time_declassify_int(int v) {
  573|  2.55k|  static_assert(sizeof(uint32_t) == sizeof(int),
  574|  2.55k|                "int is not the same size as uint32_t");
  575|       |  // See comment above.
  576|  2.55k|  CONSTTIME_DECLASSIFY(&v, sizeof(v));
  577|  2.55k|  return value_barrier_u32(v);
  578|  2.55k|}
bcm.c:value_barrier_u32:
  348|  2.55k|static inline uint32_t value_barrier_u32(uint32_t a) {
  349|  2.55k|#if defined(__GNUC__) || defined(__clang__)
  350|  2.55k|  __asm__("" : "+r"(a) : /* no inputs */);
  351|  2.55k|#endif
  352|  2.55k|  return a;
  353|  2.55k|}
bcm.c:CRYPTO_is_BMI1_capable:
 1352|     69|OPENSSL_INLINE int CRYPTO_is_BMI1_capable(void) {
 1353|       |#if defined(__BMI1__)
 1354|       |  return 1;
 1355|       |#else
 1356|     69|  return (OPENSSL_ia32cap_get()[2] & (1 << 3)) != 0;
 1357|     69|#endif
 1358|     69|}
bcm.c:CRYPTO_is_BMI2_capable:
 1368|     69|OPENSSL_INLINE int CRYPTO_is_BMI2_capable(void) {
 1369|       |#if defined(__BMI2__)
 1370|       |  return 1;
 1371|       |#else
 1372|     69|  return (OPENSSL_ia32cap_get()[2] & (1 << 8)) != 0;
 1373|     69|#endif
 1374|     69|}
bcm.c:CRYPTO_is_ADX_capable:
 1376|     69|OPENSSL_INLINE int CRYPTO_is_ADX_capable(void) {
 1377|       |#if defined(__ADX__)
 1378|       |  return 1;
 1379|       |#else
 1380|     69|  return (OPENSSL_ia32cap_get()[2] & (1 << 19)) != 0;
 1381|     69|#endif
 1382|     69|}
bcm.c:align_pointer:
  282|      8|static inline void *align_pointer(void *ptr, size_t alignment) {
  283|       |  // |alignment| must be a power of two.
  284|      8|  assert(alignment != 0 && (alignment & (alignment - 1)) == 0);
  285|       |  // Instead of aligning |ptr| as a |uintptr_t| and casting back, compute the
  286|       |  // offset and advance in pointer space. C guarantees that casting from pointer
  287|       |  // to |uintptr_t| and back gives the same pointer, but general
  288|       |  // integer-to-pointer conversions are implementation-defined. GCC does define
  289|       |  // it in the useful way, but this makes fewer assumptions.
  290|      8|  uintptr_t offset = (0u - (uintptr_t)ptr) & (alignment - 1);
  291|      8|  ptr = (char *)ptr + offset;
  292|      8|  assert(((uintptr_t)ptr & (alignment - 1)) == 0);
  293|      8|  return ptr;
  294|      8|}
bcm.c:constant_time_lt_w:
  374|  2.39M|                                               crypto_word_t b) {
  375|       |  // Consider the two cases of the problem:
  376|       |  //   msb(a) == msb(b): a < b iff the MSB of a - b is set.
  377|       |  //   msb(a) != msb(b): a < b iff the MSB of b is set.
  378|       |  //
  379|       |  // If msb(a) == msb(b) then the following evaluates as:
  380|       |  //   msb(a^((a^b)|((a-b)^a))) ==
  381|       |  //   msb(a^((a-b) ^ a))       ==   (because msb(a^b) == 0)
  382|       |  //   msb(a^a^(a-b))           ==   (rearranging)
  383|       |  //   msb(a-b)                      (because ∀x. x^x == 0)
  384|       |  //
  385|       |  // Else, if msb(a) != msb(b) then the following evaluates as:
  386|       |  //   msb(a^((a^b)|((a-b)^a))) ==
  387|       |  //   msb(a^(𝟙 | ((a-b)^a)))   ==   (because msb(a^b) == 1 and 𝟙
  388|       |  //                                  represents a value s.t. msb(𝟙) = 1)
  389|       |  //   msb(a^𝟙)                 ==   (because ORing with 1 results in 1)
  390|       |  //   msb(b)
  391|       |  //
  392|       |  //
  393|       |  // Here is an SMT-LIB verification of this formula:
  394|       |  //
  395|       |  // (define-fun lt ((a (_ BitVec 32)) (b (_ BitVec 32))) (_ BitVec 32)
  396|       |  //   (bvxor a (bvor (bvxor a b) (bvxor (bvsub a b) a)))
  397|       |  // )
  398|       |  //
  399|       |  // (declare-fun a () (_ BitVec 32))
  400|       |  // (declare-fun b () (_ BitVec 32))
  401|       |  //
  402|       |  // (assert (not (= (= #x00000001 (bvlshr (lt a b) #x0000001f)) (bvult a b))))
  403|       |  // (check-sat)
  404|       |  // (get-model)
  405|  2.39M|  return constant_time_msb_w(a^((a^b)|((a-b)^a)));
  406|  2.39M|}
mem.c:OPENSSL_memset:
 1055|   154k|static inline void *OPENSSL_memset(void *dst, int c, size_t n) {
 1056|   154k|  if (n == 0) {
  ------------------
  |  Branch (1056:7): [True: 0, False: 154k]
  ------------------
 1057|      0|    return dst;
 1058|      0|  }
 1059|       |
 1060|   154k|  return memset(dst, c, n);
 1061|   154k|}
stack.c:OPENSSL_memset:
 1055|  5.55k|static inline void *OPENSSL_memset(void *dst, int c, size_t n) {
 1056|  5.55k|  if (n == 0) {
  ------------------
  |  Branch (1056:7): [True: 0, False: 5.55k]
  ------------------
 1057|      0|    return dst;
 1058|      0|  }
 1059|       |
 1060|  5.55k|  return memset(dst, c, n);
 1061|  5.55k|}

OPENSSL_malloc:
  228|   153k|void *OPENSSL_malloc(size_t size) {
  229|   153k|  if (should_fail_allocation()) {
  ------------------
  |  Branch (229:7): [True: 0, False: 153k]
  ------------------
  230|      0|    goto err;
  231|      0|  }
  232|       |
  233|   153k|  if (OPENSSL_memory_alloc != NULL) {
  ------------------
  |  Branch (233:7): [True: 0, False: 153k]
  ------------------
  234|      0|    assert(OPENSSL_memory_free != NULL);
  235|      0|    assert(OPENSSL_memory_get_size != NULL);
  236|      0|    void *ptr = OPENSSL_memory_alloc(size);
  237|      0|    if (ptr == NULL && size != 0) {
  ------------------
  |  Branch (237:9): [True: 0, False: 0]
  |  Branch (237:24): [True: 0, False: 0]
  ------------------
  238|      0|      goto err;
  239|      0|    }
  240|      0|    return ptr;
  241|      0|  }
  242|       |
  243|   153k|  if (size + OPENSSL_MALLOC_PREFIX < size) {
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  |  Branch (243:7): [True: 0, False: 153k]
  ------------------
  244|       |    // |OPENSSL_malloc| is a central function in BoringSSL thus a reference to
  245|       |    // |kBoringSSLBinaryTag| is created here so that the tag isn't discarded by
  246|       |    // the linker. The following is sufficient to stop GCC, Clang, and MSVC
  247|       |    // optimising away the reference at the time of writing. Since this
  248|       |    // probably results in an actual memory reference, it is put in this very
  249|       |    // rare code path.
  250|      0|    uint8_t unused = *(volatile uint8_t *)kBoringSSLBinaryTag;
  251|      0|    (void) unused;
  252|      0|    goto err;
  253|      0|  }
  254|       |
  255|   153k|  void *ptr = malloc(size + OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  256|   153k|  if (ptr == NULL) {
  ------------------
  |  Branch (256:7): [True: 0, False: 153k]
  ------------------
  257|      0|    goto err;
  258|      0|  }
  259|       |
  260|   153k|  *(size_t *)ptr = size;
  261|       |
  262|   153k|  __asan_poison_memory_region(ptr, OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  263|   153k|  return ((uint8_t *)ptr) + OPENSSL_MALLOC_PREFIX;
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  264|       |
  265|      0| err:
  266|       |  // This only works because ERR does not call OPENSSL_malloc.
  267|      0|  OPENSSL_PUT_ERROR(CRYPTO, ERR_R_MALLOC_FAILURE);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  268|      0|  return NULL;
  269|   153k|}
OPENSSL_free:
  271|   212k|void OPENSSL_free(void *orig_ptr) {
  272|   212k|  if (orig_ptr == NULL) {
  ------------------
  |  Branch (272:7): [True: 59.0k, False: 153k]
  ------------------
  273|  59.0k|    return;
  274|  59.0k|  }
  275|       |
  276|   153k|  if (OPENSSL_memory_free != NULL) {
  ------------------
  |  Branch (276:7): [True: 0, False: 153k]
  ------------------
  277|      0|    OPENSSL_memory_free(orig_ptr);
  278|      0|    return;
  279|      0|  }
  280|       |
  281|   153k|  void *ptr = ((uint8_t *)orig_ptr) - OPENSSL_MALLOC_PREFIX;
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  282|   153k|  __asan_unpoison_memory_region(ptr, OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  283|       |
  284|   153k|  size_t size = *(size_t *)ptr;
  285|   153k|  OPENSSL_cleanse(ptr, size + OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|   153k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  286|       |
  287|       |// ASan knows to intercept malloc and free, but not sdallocx.
  288|       |#if defined(OPENSSL_ASAN)
  289|       |  (void)sdallocx;
  290|       |  free(ptr);
  291|       |#else
  292|   153k|  if (sdallocx) {
  ------------------
  |  Branch (292:7): [True: 0, False: 153k]
  ------------------
  293|      0|    sdallocx(ptr, size + OPENSSL_MALLOC_PREFIX, 0 /* flags */);
  ------------------
  |  |   83|      0|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  294|   153k|  } else {
  295|   153k|    free(ptr);
  296|   153k|  }
  297|   153k|#endif
  298|   153k|}
OPENSSL_realloc:
  300|  8.13k|void *OPENSSL_realloc(void *orig_ptr, size_t new_size) {
  301|  8.13k|  if (orig_ptr == NULL) {
  ------------------
  |  Branch (301:7): [True: 2.77k, False: 5.35k]
  ------------------
  302|  2.77k|    return OPENSSL_malloc(new_size);
  303|  2.77k|  }
  304|       |
  305|  5.35k|  size_t old_size;
  306|  5.35k|  if (OPENSSL_memory_get_size != NULL) {
  ------------------
  |  Branch (306:7): [True: 0, False: 5.35k]
  ------------------
  307|      0|    old_size = OPENSSL_memory_get_size(orig_ptr);
  308|  5.35k|  } else {
  309|  5.35k|    void *ptr = ((uint8_t *)orig_ptr) - OPENSSL_MALLOC_PREFIX;
  ------------------
  |  |   83|  5.35k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  310|  5.35k|    __asan_unpoison_memory_region(ptr, OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|  5.35k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  311|  5.35k|    old_size = *(size_t *)ptr;
  312|  5.35k|    __asan_poison_memory_region(ptr, OPENSSL_MALLOC_PREFIX);
  ------------------
  |  |   83|  5.35k|#define OPENSSL_MALLOC_PREFIX 8
  ------------------
  313|  5.35k|  }
  314|       |
  315|  5.35k|  void *ret = OPENSSL_malloc(new_size);
  316|  5.35k|  if (ret == NULL) {
  ------------------
  |  Branch (316:7): [True: 0, False: 5.35k]
  ------------------
  317|      0|    return NULL;
  318|      0|  }
  319|       |
  320|  5.35k|  size_t to_copy = new_size;
  321|  5.35k|  if (old_size < to_copy) {
  ------------------
  |  Branch (321:7): [True: 5.35k, False: 0]
  ------------------
  322|  5.35k|    to_copy = old_size;
  323|  5.35k|  }
  324|       |
  325|  5.35k|  memcpy(ret, orig_ptr, to_copy);
  326|  5.35k|  OPENSSL_free(orig_ptr);
  327|       |
  328|  5.35k|  return ret;
  329|  5.35k|}
OPENSSL_cleanse:
  331|   154k|void OPENSSL_cleanse(void *ptr, size_t len) {
  332|       |#if defined(OPENSSL_WINDOWS)
  333|       |  SecureZeroMemory(ptr, len);
  334|       |#else
  335|   154k|  OPENSSL_memset(ptr, 0, len);
  336|       |
  337|   154k|#if !defined(OPENSSL_NO_ASM)
  338|       |  /* As best as we can tell, this is sufficient to break any optimisations that
  339|       |     might try to eliminate "superfluous" memsets. If there's an easy way to
  340|       |     detect memset_s, it would be better to use that. */
  341|   154k|  __asm__ __volatile__("" : : "r"(ptr) : "memory");
  342|   154k|#endif
  343|   154k|#endif  // !OPENSSL_NO_ASM
  344|   154k|}
mem.c:should_fail_allocation:
  225|   153k|static int should_fail_allocation(void) { return 0; }
mem.c:__asan_poison_memory_region:
   90|   158k|static void __asan_poison_memory_region(const void *addr, size_t size) {}
mem.c:__asan_unpoison_memory_region:
   91|   158k|static void __asan_unpoison_memory_region(const void *addr, size_t size) {}

sk_new:
   72|  2.77k|_STACK *sk_new(OPENSSL_sk_cmp_func comp) {
   73|  2.77k|  _STACK *ret = OPENSSL_malloc(sizeof(_STACK));
   74|  2.77k|  if (ret == NULL) {
  ------------------
  |  Branch (74:7): [True: 0, False: 2.77k]
  ------------------
   75|      0|    return NULL;
   76|      0|  }
   77|  2.77k|  OPENSSL_memset(ret, 0, sizeof(_STACK));
   78|       |
   79|  2.77k|  ret->data = OPENSSL_malloc(sizeof(void *) * kMinSize);
   80|  2.77k|  if (ret->data == NULL) {
  ------------------
  |  Branch (80:7): [True: 0, False: 2.77k]
  ------------------
   81|      0|    goto err;
   82|      0|  }
   83|       |
   84|  2.77k|  OPENSSL_memset(ret->data, 0, sizeof(void *) * kMinSize);
   85|       |
   86|  2.77k|  ret->comp = comp;
   87|  2.77k|  ret->num_alloc = kMinSize;
   88|       |
   89|  2.77k|  return ret;
   90|       |
   91|      0|err:
   92|      0|  OPENSSL_free(ret);
   93|      0|  return NULL;
   94|  2.77k|}
sk_new_null:
   96|  2.77k|_STACK *sk_new_null(void) { return sk_new(NULL); }
sk_num:
   98|  5.12M|size_t sk_num(const _STACK *sk) {
   99|  5.12M|  if (sk == NULL) {
  ------------------
  |  Branch (99:7): [True: 0, False: 5.12M]
  ------------------
  100|      0|    return 0;
  101|      0|  }
  102|  5.12M|  return sk->num;
  103|  5.12M|}
sk_value:
  114|  5.12M|void *sk_value(const _STACK *sk, size_t i) {
  115|  5.12M|  if (!sk || i >= sk->num) {
  ------------------
  |  Branch (115:7): [True: 0, False: 5.12M]
  |  Branch (115:14): [True: 0, False: 5.12M]
  ------------------
  116|      0|    return NULL;
  117|      0|  }
  118|  5.12M|  return sk->data[i];
  119|  5.12M|}
sk_free:
  128|  2.77k|void sk_free(_STACK *sk) {
  129|  2.77k|  if (sk == NULL) {
  ------------------
  |  Branch (129:7): [True: 0, False: 2.77k]
  ------------------
  130|      0|    return;
  131|      0|  }
  132|  2.77k|  OPENSSL_free(sk->data);
  133|  2.77k|  OPENSSL_free(sk);
  134|  2.77k|}
sk_pop_free_ex:
  137|  2.77k|                    OPENSSL_sk_free_func free_func) {
  138|  2.77k|  if (sk == NULL) {
  ------------------
  |  Branch (138:7): [True: 0, False: 2.77k]
  ------------------
  139|      0|    return;
  140|      0|  }
  141|       |
  142|  33.5k|  for (size_t i = 0; i < sk->num; i++) {
  ------------------
  |  Branch (142:22): [True: 30.7k, False: 2.77k]
  ------------------
  143|  30.7k|    if (sk->data[i] != NULL) {
  ------------------
  |  Branch (143:9): [True: 30.7k, False: 0]
  ------------------
  144|  30.7k|      call_free_func(free_func, sk->data[i]);
  145|  30.7k|    }
  146|  30.7k|  }
  147|  2.77k|  sk_free(sk);
  148|  2.77k|}
sk_insert:
  161|  30.7k|size_t sk_insert(_STACK *sk, void *p, size_t where) {
  162|  30.7k|  if (sk == NULL) {
  ------------------
  |  Branch (162:7): [True: 0, False: 30.7k]
  ------------------
  163|      0|    return 0;
  164|      0|  }
  165|       |
  166|  30.7k|  if (sk->num >= INT_MAX) {
  ------------------
  |  Branch (166:7): [True: 0, False: 30.7k]
  ------------------
  167|      0|    OPENSSL_PUT_ERROR(CRYPTO, ERR_R_OVERFLOW);
  ------------------
  |  |  441|      0|  ERR_put_error(ERR_LIB_##library, 0, reason, __FILE__, __LINE__)
  ------------------
  168|      0|    return 0;
  169|      0|  }
  170|       |
  171|  30.7k|  if (sk->num_alloc <= sk->num + 1) {
  ------------------
  |  Branch (171:7): [True: 5.35k, False: 25.3k]
  ------------------
  172|       |    // Attempt to double the size of the array.
  173|  5.35k|    size_t new_alloc = sk->num_alloc << 1;
  174|  5.35k|    size_t alloc_size = new_alloc * sizeof(void *);
  175|  5.35k|    void **data;
  176|       |
  177|       |    // If the doubling overflowed, try to increment.
  178|  5.35k|    if (new_alloc < sk->num_alloc || alloc_size / sizeof(void *) != new_alloc) {
  ------------------
  |  Branch (178:9): [True: 0, False: 5.35k]
  |  Branch (178:38): [True: 0, False: 5.35k]
  ------------------
  179|      0|      new_alloc = sk->num_alloc + 1;
  180|      0|      alloc_size = new_alloc * sizeof(void *);
  181|      0|    }
  182|       |
  183|       |    // If the increment also overflowed, fail.
  184|  5.35k|    if (new_alloc < sk->num_alloc || alloc_size / sizeof(void *) != new_alloc) {
  ------------------
  |  Branch (184:9): [True: 0, False: 5.35k]
  |  Branch (184:38): [True: 0, False: 5.35k]
  ------------------
  185|      0|      return 0;
  186|      0|    }
  187|       |
  188|  5.35k|    data = OPENSSL_realloc(sk->data, alloc_size);
  189|  5.35k|    if (data == NULL) {
  ------------------
  |  Branch (189:9): [True: 0, False: 5.35k]
  ------------------
  190|      0|      return 0;
  191|      0|    }
  192|       |
  193|  5.35k|    sk->data = data;
  194|  5.35k|    sk->num_alloc = new_alloc;
  195|  5.35k|  }
  196|       |
  197|  30.7k|  if (where >= sk->num) {
  ------------------
  |  Branch (197:7): [True: 30.7k, False: 0]
  ------------------
  198|  30.7k|    sk->data[sk->num] = p;
  199|  30.7k|  } else {
  200|      0|    OPENSSL_memmove(&sk->data[where + 1], &sk->data[where],
  201|      0|                    sizeof(void *) * (sk->num - where));
  202|      0|    sk->data[where] = p;
  203|      0|  }
  204|       |
  205|  30.7k|  sk->num++;
  206|  30.7k|  sk->sorted = 0;
  207|       |
  208|  30.7k|  return sk->num;
  209|  30.7k|}
sk_push:
  340|  30.7k|size_t sk_push(_STACK *sk, void *p) { return (sk_insert(sk, p, sk->num)); }

_Z7mod_expP9bignum_stPKS_S2_S2_P10bignum_ctx:
   29|  2.77k|            BN_CTX *ctx) {
   30|  2.77k|  if (BN_is_one(m)) {
  ------------------
  |  Branch (30:7): [True: 37, False: 2.74k]
  ------------------
   31|     37|    BN_zero(r);
   32|     37|    return 1;
   33|     37|  }
   34|       |
   35|  2.74k|  bssl::UniquePtr<BIGNUM> exp(BN_dup(p));
   36|  2.74k|  bssl::UniquePtr<BIGNUM> base(BN_new());
   37|  2.74k|  if (!exp || !base) {
  ------------------
  |  Branch (37:7): [True: 0, False: 2.74k]
  |  Branch (37:15): [True: 0, False: 2.74k]
  ------------------
   38|      0|    return 0;
   39|      0|  }
   40|  2.74k|  if (!BN_one(r) || !BN_nnmod(base.get(), a, m, ctx)) {
  ------------------
  |  Branch (40:7): [True: 0, False: 2.74k]
  |  Branch (40:21): [True: 0, False: 2.74k]
  ------------------
   41|      0|    return 0;
   42|      0|  }
   43|       |
   44|   441k|  while (!BN_is_zero(exp.get())) {
  ------------------
  |  Branch (44:10): [True: 438k, False: 2.74k]
  ------------------
   45|   438k|    if (BN_is_odd(exp.get())) {
  ------------------
  |  Branch (45:9): [True: 175k, False: 262k]
  ------------------
   46|   175k|      if (!BN_mul(r, r, base.get(), ctx) || !BN_nnmod(r, r, m, ctx)) {
  ------------------
  |  Branch (46:11): [True: 0, False: 175k]
  |  Branch (46:45): [True: 0, False: 175k]
  ------------------
   47|      0|        return 0;
   48|      0|      }
   49|   175k|    }
   50|   438k|    if (!BN_rshift1(exp.get(), exp.get()) ||
  ------------------
  |  Branch (50:9): [True: 0, False: 438k]
  ------------------
   51|   438k|        !BN_mul(base.get(), base.get(), base.get(), ctx) ||
  ------------------
  |  Branch (51:9): [True: 0, False: 438k]
  ------------------
   52|   438k|        !BN_nnmod(base.get(), base.get(), m, ctx)) {
  ------------------
  |  Branch (52:9): [True: 0, False: 438k]
  ------------------
   53|      0|      return 0;
   54|      0|    }
   55|   438k|  }
   56|       |
   57|  2.74k|  return 1;
   58|  2.74k|}
LLVMFuzzerTestOneInput:
   60|  2.89k|extern "C" int LLVMFuzzerTestOneInput(const uint8_t *buf, size_t len) {
   61|  2.89k|  CBS cbs, child0, child1, child2;
   62|  2.89k|  uint8_t sign;
   63|  2.89k|  CBS_init(&cbs, buf, len);
   64|  2.89k|  if (!CBS_get_u16_length_prefixed(&cbs, &child0) ||
  ------------------
  |  Branch (64:7): [True: 19, False: 2.87k]
  ------------------
   65|  2.89k|      !CBS_get_u8(&child0, &sign) ||
  ------------------
  |  Branch (65:7): [True: 20, False: 2.85k]
  ------------------
   66|  2.89k|      CBS_len(&child0) == 0 ||
  ------------------
  |  Branch (66:7): [True: 1, False: 2.85k]
  ------------------
   67|  2.89k|      !CBS_get_u16_length_prefixed(&cbs, &child1) ||
  ------------------
  |  Branch (67:7): [True: 22, False: 2.83k]
  ------------------
   68|  2.89k|      CBS_len(&child1) == 0 ||
  ------------------
  |  Branch (68:7): [True: 8, False: 2.82k]
  ------------------
   69|  2.89k|      !CBS_get_u16_length_prefixed(&cbs, &child2) ||
  ------------------
  |  Branch (69:7): [True: 19, False: 2.80k]
  ------------------
   70|  2.89k|      CBS_len(&child2) == 0) {
  ------------------
  |  Branch (70:7): [True: 1, False: 2.80k]
  ------------------
   71|     90|    return 0;
   72|     90|  }
   73|       |
   74|       |  // Don't fuzz inputs larger than 512 bytes (4096 bits). This isn't ideal, but
   75|       |  // the naive |mod_exp| above is somewhat slow, so this otherwise causes the
   76|       |  // fuzzers to spend a lot of time exploring timeouts.
   77|  2.80k|  if (CBS_len(&child0) > 512 ||
  ------------------
  |  Branch (77:7): [True: 2, False: 2.80k]
  ------------------
   78|  2.80k|      CBS_len(&child1) > 512 ||
  ------------------
  |  Branch (78:7): [True: 8, False: 2.79k]
  ------------------
   79|  2.80k|      CBS_len(&child2) > 512) {
  ------------------
  |  Branch (79:7): [True: 8, False: 2.78k]
  ------------------
   80|     18|    return 0;
   81|     18|  }
   82|       |
   83|  2.78k|  bssl::UniquePtr<BIGNUM> base(
   84|  2.78k|      BN_bin2bn(CBS_data(&child0), CBS_len(&child0), nullptr));
   85|  2.78k|  BN_set_negative(base.get(), sign % 2);
   86|  2.78k|  bssl::UniquePtr<BIGNUM> power(
   87|  2.78k|      BN_bin2bn(CBS_data(&child1), CBS_len(&child1), nullptr));
   88|  2.78k|  bssl::UniquePtr<BIGNUM> modulus(
   89|  2.78k|      BN_bin2bn(CBS_data(&child2), CBS_len(&child2), nullptr));
   90|       |
   91|  2.78k|  if (BN_is_zero(modulus.get())) {
  ------------------
  |  Branch (91:7): [True: 10, False: 2.77k]
  ------------------
   92|     10|    return 0;
   93|     10|  }
   94|       |
   95|  2.77k|  bssl::UniquePtr<BN_CTX> ctx(BN_CTX_new());
   96|  2.77k|  bssl::UniquePtr<BIGNUM> result(BN_new());
   97|  2.77k|  bssl::UniquePtr<BIGNUM> expected(BN_new());
   98|  2.77k|  CHECK(ctx);
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
   99|  2.77k|  CHECK(result);
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  100|  2.77k|  CHECK(expected);
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  101|       |
  102|  2.77k|  CHECK(mod_exp(expected.get(), base.get(), power.get(), modulus.get(),
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  103|  2.77k|                ctx.get()));
  104|  2.77k|  CHECK(BN_mod_exp(result.get(), base.get(), power.get(), modulus.get(),
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  105|  2.77k|                   ctx.get()));
  106|  2.77k|  CHECK(BN_cmp(result.get(), expected.get()) == 0);
  ------------------
  |  |   20|  2.77k|  do {                              \
  |  |   21|  2.77k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 2.77k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  2.77k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  107|       |
  108|  2.77k|  if (BN_is_odd(modulus.get())) {
  ------------------
  |  Branch (108:7): [True: 1.27k, False: 1.50k]
  ------------------
  109|  1.27k|    bssl::UniquePtr<BN_MONT_CTX> mont(
  110|  1.27k|        BN_MONT_CTX_new_for_modulus(modulus.get(), ctx.get()));
  111|  1.27k|    CHECK(mont);
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  112|       |    // |BN_mod_exp_mont| and |BN_mod_exp_mont_consttime| require reduced inputs.
  113|  1.27k|    CHECK(BN_nnmod(base.get(), base.get(), modulus.get(), ctx.get()));
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  114|  1.27k|    CHECK(BN_mod_exp_mont(result.get(), base.get(), power.get(), modulus.get(),
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  115|  1.27k|                          ctx.get(), mont.get()));
  116|  1.27k|    CHECK(BN_cmp(result.get(), expected.get()) == 0);
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  117|  1.27k|    CHECK(BN_mod_exp_mont_consttime(result.get(), base.get(), power.get(),
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  118|  1.27k|                                    modulus.get(), ctx.get(), mont.get()));
  119|  1.27k|    CHECK(BN_cmp(result.get(), expected.get()) == 0);
  ------------------
  |  |   20|  1.27k|  do {                              \
  |  |   21|  1.27k|    if (!(expr)) {                  \
  |  |  ------------------
  |  |  |  Branch (21:9): [True: 0, False: 1.27k]
  |  |  ------------------
  |  |   22|      0|      printf("%s failed\n", #expr); \
  |  |   23|      0|      abort();                      \
  |  |   24|      0|    }                               \
  |  |   25|  1.27k|  } while (false)
  |  |  ------------------
  |  |  |  Branch (25:12): [Folded - Ignored]
  |  |  ------------------
  ------------------
  120|  1.27k|  }
  121|       |
  122|  2.77k|  uint8_t *data = (uint8_t *)OPENSSL_malloc(BN_num_bytes(result.get()));
  123|  2.77k|  BN_bn2bin(result.get(), data);
  124|  2.77k|  OPENSSL_free(data);
  125|       |
  126|  2.77k|  return 0;
  127|  2.77k|}

_ZN4bssl8internal7DeleterclI9bignum_stEEvPT_:
  560|  19.4k|  void operator()(T *ptr) {
  561|       |    // Rather than specialize Deleter for each type, we specialize
  562|       |    // DeleterImpl. This allows bssl::UniquePtr<T> to be used while only
  563|       |    // including base.h as long as the destructor is not emitted. This matches
  564|       |    // std::unique_ptr's behavior on forward-declared types.
  565|       |    //
  566|       |    // DeleterImpl itself is specialized in the corresponding module's header
  567|       |    // and must be included to release an object. If not included, the compiler
  568|       |    // will error that DeleterImpl<T> does not have a method Free.
  569|  19.4k|    DeleterImpl<T>::Free(ptr);
  570|  19.4k|  }
_ZN4bssl8internal11DeleterImplI9bignum_stvE4FreeEPS2_:
  635|  19.4k|    static void Free(type *ptr) { deleter(ptr); } \
_ZN4bssl8internal7DeleterclI10bignum_ctxEEvPT_:
  560|  2.77k|  void operator()(T *ptr) {
  561|       |    // Rather than specialize Deleter for each type, we specialize
  562|       |    // DeleterImpl. This allows bssl::UniquePtr<T> to be used while only
  563|       |    // including base.h as long as the destructor is not emitted. This matches
  564|       |    // std::unique_ptr's behavior on forward-declared types.
  565|       |    //
  566|       |    // DeleterImpl itself is specialized in the corresponding module's header
  567|       |    // and must be included to release an object. If not included, the compiler
  568|       |    // will error that DeleterImpl<T> does not have a method Free.
  569|  2.77k|    DeleterImpl<T>::Free(ptr);
  570|  2.77k|  }
_ZN4bssl8internal11DeleterImplI10bignum_ctxvE4FreeEPS2_:
  635|  2.77k|    static void Free(type *ptr) { deleter(ptr); } \
_ZN4bssl8internal7DeleterclI14bn_mont_ctx_stEEvPT_:
  560|  1.27k|  void operator()(T *ptr) {
  561|       |    // Rather than specialize Deleter for each type, we specialize
  562|       |    // DeleterImpl. This allows bssl::UniquePtr<T> to be used while only
  563|       |    // including base.h as long as the destructor is not emitted. This matches
  564|       |    // std::unique_ptr's behavior on forward-declared types.
  565|       |    //
  566|       |    // DeleterImpl itself is specialized in the corresponding module's header
  567|       |    // and must be included to release an object. If not included, the compiler
  568|       |    // will error that DeleterImpl<T> does not have a method Free.
  569|  1.27k|    DeleterImpl<T>::Free(ptr);
  570|  1.27k|  }
_ZN4bssl8internal11DeleterImplI14bn_mont_ctx_stvE4FreeEPS2_:
  635|  1.27k|    static void Free(type *ptr) { deleter(ptr); } \

bcm.c:sk_BIGNUM_pop_free:
  447|  2.77k|                                           sk_##name##_free_func free_func) { \
  448|  2.77k|    sk_pop_free_ex((_STACK *)sk, sk_##name##_call_free_func,                  \
  449|  2.77k|                   (OPENSSL_sk_free_func)free_func);                          \
  450|  2.77k|  }                                                                           \
bcm.c:sk_BIGNUM_call_free_func:
  391|  30.7k|      OPENSSL_sk_free_func free_func, void *ptr) {                            \
  392|  30.7k|    ((sk_##name##_free_func)free_func)((ptrtype)ptr);                         \
  393|  30.7k|  }                                                                           \
bcm.c:sk_BIGNUM_new_null:
  420|  2.77k|  OPENSSL_INLINE STACK_OF(name) *sk_##name##_new_null(void) {                 \
  421|  2.77k|    return (STACK_OF(name) *)sk_new_null();                                   \
  422|  2.77k|  }                                                                           \
bcm.c:sk_BIGNUM_num:
  424|  5.12M|  OPENSSL_INLINE size_t sk_##name##_num(const STACK_OF(name) *sk) {           \
  425|  5.12M|    return sk_num((const _STACK *)sk);                                        \
  426|  5.12M|  }                                                                           \
bcm.c:sk_BIGNUM_push:
  483|  30.7k|  OPENSSL_INLINE size_t sk_##name##_push(STACK_OF(name) *sk, ptrtype p) {     \
  484|  30.7k|    return sk_push((_STACK *)sk, (void *)p);                                  \
  485|  30.7k|  }                                                                           \
bcm.c:sk_BIGNUM_value:
  433|  5.12M|                                           size_t i) {                        \
  434|  5.12M|    return (ptrtype)sk_value((const _STACK *)sk, i);                          \
  435|  5.12M|  }                                                                           \

