meshopt_decodeIndexBuffer: 385| 4.31k|{ 386| 4.31k| using namespace meshopt; 387| | 388| 4.31k| assert(index_count % 3 == 0); ------------------ | Branch (388:2): [True: 4.31k, False: 0] ------------------ 389| 4.31k| assert(index_size == 2 || index_size == 4); ------------------ | Branch (389:2): [True: 2.15k, False: 2.15k] | Branch (389:2): [True: 2.15k, False: 0] | Branch (389:2): [True: 4.31k, False: 0] ------------------ 390| | 391| | // the minimum valid encoding is header, 1 byte per triangle and a 16-byte codeaux table 392| 4.31k| if (buffer_size < 1 + index_count / 3 + 16) ------------------ | Branch (392:6): [True: 1.86k, False: 2.45k] ------------------ 393| 1.86k| return -2; 394| | 395| 2.45k| if ((buffer[0] & 0xf0) != kIndexHeader) ------------------ | Branch (395:6): [True: 1.77k, False: 674] ------------------ 396| 1.77k| return -1; 397| | 398| 674| int version = buffer[0] & 0x0f; 399| 674| if (version > kDecodeIndexVersion) ------------------ | Branch (399:6): [True: 70, False: 604] ------------------ 400| 70| return -1; 401| | 402| 604| EdgeFifo edgefifo; 403| 604| memset(edgefifo, -1, sizeof(edgefifo)); 404| | 405| 604| VertexFifo vertexfifo; 406| 604| memset(vertexfifo, -1, sizeof(vertexfifo)); 407| | 408| 604| size_t edgefifooffset = 0; 409| 604| size_t vertexfifooffset = 0; 410| | 411| 604| unsigned int next = 0; 412| 604| unsigned int last = 0; 413| | 414| 604| int fecmax = version >= 1 ? 13 : 15; ------------------ | Branch (414:15): [True: 104, False: 500] ------------------ 415| | 416| | // since we store 16-byte codeaux table at the end, triangle data has to begin before data_safe_end 417| 604| const unsigned char* code = buffer + 1; 418| 604| const unsigned char* code_end = code + index_count / 3; 419| 604| const unsigned char* data = code_end; 420| | 421| | // each triangle reads at most 16 bytes of data: 1b for codeaux and 5b for each free index 422| 604| const unsigned char* data_safe_end = buffer + buffer_size - 16; 423| | 424| 604| const unsigned char* codeaux_table = data_safe_end; 425| | 426| 10.8k| while (code < code_end) ------------------ | Branch (426:9): [True: 10.4k, False: 346] ------------------ 427| 10.4k| { 428| 10.4k| unsigned char codetri = *code++; 429| | 430| 10.4k| if (codetri < 0xf0) ------------------ | Branch (430:7): [True: 6.29k, False: 4.18k] ------------------ 431| 6.29k| { 432| 6.29k| int fe = codetri >> 4; 433| | 434| | // fifo reads are wrapped around 16 entry buffer 435| 6.29k| unsigned int a = edgefifo[(edgefifooffset - 1 - fe) & 15][0]; 436| 6.29k| unsigned int b = edgefifo[(edgefifooffset - 1 - fe) & 15][1]; 437| 6.29k| unsigned int c = 0; 438| | 439| 6.29k| int fec = codetri & 15; 440| | 441| | // note: this is the most common path in the entire decoder 442| | // inside this if we try to stay branchless (by using cmov/etc.) since these aren't predictable 443| 6.29k| if (fec < fecmax) ------------------ | Branch (443:8): [True: 5.20k, False: 1.08k] ------------------ 444| 5.20k| { 445| | // fifo reads are wrapped around 16 entry buffer 446| 5.20k| unsigned int cf = vertexfifo[(vertexfifooffset - 1 - fec) & 15]; 447| | 448| 5.20k|#if (defined(__GNUC__) || defined(__clang__)) && (defined(__x86_64__) || defined(__aarch64__)) 449| | // on clang, x86-cmov-conversion pass emits a branch since cf is a load from memory; asm barrier defeats this 450| | // this can be fixed with __builtin_unpredictable, but gcc doesn't support it and has a similar problem due to if-conversion 451| | // additionally, gcc can fuse a/b into one 64-bit load but use 32-bit edgefifo[] stores, which breaks load store forwarding 452| 5.20k| __asm__("" : "+r"(a) : "r"(cf)); 453| 5.20k|#endif 454| | 455| 5.20k| c = (fec == 0) ? next : cf; ------------------ | Branch (455:9): [True: 2.08k, False: 3.12k] ------------------ 456| | 457| 5.20k| int fec0 = fec == 0; 458| 5.20k| next += fec0; 459| | 460| | // push vertex fifo must match the encoding step *exactly* otherwise the data will not be decoded correctly 461| 5.20k| pushVertexFifo(vertexfifo, c, vertexfifooffset, fec0); 462| 5.20k| } 463| 1.08k| else 464| 1.08k| { 465| | // make sure we have enough data to read for a triangle; this check covers worst case advance 466| 1.08k| if (data > data_safe_end) ------------------ | Branch (466:9): [True: 56, False: 1.02k] ------------------ 467| 56| return -2; 468| | 469| | // fec * 2 - 27 decodes 13, 14 into -1, 1 470| | // note that we need to update the last index since free indices are delta-encoded 471| 1.02k| last = c = (fec != 15) ? last + (fec * 2 - 27) : decodeIndex(data, last); ------------------ | Branch (471:16): [True: 142, False: 884] ------------------ 472| | 473| | // push vertex/edge fifo must match the encoding step *exactly* otherwise the data will not be decoded correctly 474| 1.02k| pushVertexFifo(vertexfifo, c, vertexfifooffset); 475| 1.02k| } 476| | 477| | // push edge fifo must match the encoding step *exactly* otherwise the data will not be decoded correctly 478| 6.23k| pushEdgeFifo(edgefifo, c, b, edgefifooffset); 479| 6.23k| pushEdgeFifo(edgefifo, a, c, edgefifooffset); 480| | 481| | // output triangle 482| 6.23k| destination = writeTriangle(destination, index_size, a, b, c); 483| 6.23k| } 484| 4.18k| else 485| 4.18k| { 486| | // fast path: read codeaux from the table 487| 4.18k| if (codetri < 0xfe) ------------------ | Branch (487:8): [True: 1.37k, False: 2.81k] ------------------ 488| 1.37k| { 489| 1.37k| unsigned char codeaux = codeaux_table[codetri & 15]; 490| | 491| | // note: table can't contain feb/fec=15 492| 1.37k| int feb = codeaux >> 4; 493| 1.37k| int fec = codeaux & 15; 494| | 495| | // fifo reads are wrapped around 16 entry buffer 496| | // also note that we increment next for all three vertices before decoding indices - this matches encoder behavior 497| 1.37k| unsigned int a = next++; 498| | 499| 1.37k| unsigned int bf = vertexfifo[(vertexfifooffset - feb) & 15]; 500| 1.37k| unsigned int b = (feb == 0) ? next : bf; ------------------ | Branch (500:22): [True: 382, False: 992] ------------------ 501| | 502| 1.37k| int feb0 = feb == 0; 503| 1.37k| next += feb0; 504| | 505| 1.37k| unsigned int cf = vertexfifo[(vertexfifooffset - fec) & 15]; 506| 1.37k| unsigned int c = (fec == 0) ? next : cf; ------------------ | Branch (506:22): [True: 398, False: 976] ------------------ 507| | 508| 1.37k| int fec0 = fec == 0; 509| 1.37k| next += fec0; 510| | 511| | // output triangle 512| 1.37k| destination = writeTriangle(destination, index_size, a, b, c); 513| | 514| | // push vertex/edge fifo must match the encoding step *exactly* otherwise the data will not be decoded correctly 515| 1.37k| pushVertexFifo(vertexfifo, a, vertexfifooffset); 516| 1.37k| pushVertexFifo(vertexfifo, b, vertexfifooffset, feb0); 517| 1.37k| pushVertexFifo(vertexfifo, c, vertexfifooffset, fec0); 518| | 519| 1.37k| pushEdgeFifo(edgefifo, b, a, edgefifooffset); 520| 1.37k| pushEdgeFifo(edgefifo, c, b, edgefifooffset); 521| 1.37k| pushEdgeFifo(edgefifo, a, c, edgefifooffset); 522| 1.37k| } 523| 2.81k| else 524| 2.81k| { 525| | // make sure we have enough data to read for a triangle; this check covers worst case advance 526| 2.81k| if (data > data_safe_end) ------------------ | Branch (526:9): [True: 202, False: 2.61k] ------------------ 527| 202| return -2; 528| | 529| | // slow path: read a full byte for codeaux instead of using a table lookup 530| 2.61k| unsigned char codeaux = *data++; 531| | 532| 2.61k| int fea = codetri == 0xfe ? 0 : 15; ------------------ | Branch (532:15): [True: 906, False: 1.70k] ------------------ 533| 2.61k| int feb = codeaux >> 4; 534| 2.61k| int fec = codeaux & 15; 535| | 536| | // reset: codeaux is 0 but encoded as not-a-table 537| 2.61k| if (codeaux == 0) ------------------ | Branch (537:9): [True: 510, False: 2.10k] ------------------ 538| 510| next = 0; 539| | 540| | // fifo reads are wrapped around 16 entry buffer 541| | // also note that we increment next for all three vertices before decoding indices - this matches encoder behavior 542| 2.61k| unsigned int a = (fea == 0) ? next++ : 0; ------------------ | Branch (542:22): [True: 906, False: 1.70k] ------------------ 543| 2.61k| unsigned int b = (feb == 0) ? next++ : vertexfifo[(vertexfifooffset - feb) & 15]; ------------------ | Branch (543:22): [True: 668, False: 1.94k] ------------------ 544| 2.61k| unsigned int c = (fec == 0) ? next++ : vertexfifo[(vertexfifooffset - fec) & 15]; ------------------ | Branch (544:22): [True: 682, False: 1.93k] ------------------ 545| | 546| | // note that we need to update the last index since free indices are delta-encoded 547| 2.61k| if (fea == 15) ------------------ | Branch (547:9): [True: 1.70k, False: 906] ------------------ 548| 1.70k| last = a = decodeIndex(data, last); 549| | 550| 2.61k| if (feb == 15) ------------------ | Branch (550:9): [True: 864, False: 1.74k] ------------------ 551| 864| last = b = decodeIndex(data, last); 552| | 553| 2.61k| if (fec == 15) ------------------ | Branch (553:9): [True: 842, False: 1.77k] ------------------ 554| 842| last = c = decodeIndex(data, last); 555| | 556| | // output triangle 557| 2.61k| destination = writeTriangle(destination, index_size, a, b, c); 558| | 559| | // push vertex/edge fifo must match the encoding step *exactly* otherwise the data will not be decoded correctly 560| 2.61k| pushVertexFifo(vertexfifo, a, vertexfifooffset); 561| 2.61k| pushVertexFifo(vertexfifo, b, vertexfifooffset, (feb == 0) | (feb == 15)); 562| 2.61k| pushVertexFifo(vertexfifo, c, vertexfifooffset, (fec == 0) | (fec == 15)); 563| | 564| 2.61k| pushEdgeFifo(edgefifo, b, a, edgefifooffset); 565| 2.61k| pushEdgeFifo(edgefifo, c, b, edgefifooffset); 566| 2.61k| pushEdgeFifo(edgefifo, a, c, edgefifooffset); 567| 2.61k| } 568| 4.18k| } 569| 10.4k| } 570| | 571| | // we should've read all data bytes and stopped at the boundary between data and codeaux table 572| 346| if (data != data_safe_end) ------------------ | Branch (572:6): [True: 334, False: 12] ------------------ 573| 334| return -3; 574| | 575| 12| return 0; 576| 346|} meshopt_decodeIndexSequence: 648| 4.31k|{ 649| 4.31k| using namespace meshopt; 650| | 651| | // the minimum valid encoding is header, 1 byte per index and a 4-byte tail 652| 4.31k| if (buffer_size < 1 + index_count + 4) ------------------ | Branch (652:6): [True: 2.44k, False: 1.87k] ------------------ 653| 2.44k| return -2; 654| | 655| 1.87k| if ((buffer[0] & 0xf0) != kSequenceHeader) ------------------ | Branch (655:6): [True: 1.51k, False: 358] ------------------ 656| 1.51k| return -1; 657| | 658| 358| int version = buffer[0] & 0x0f; 659| 358| if (version > kDecodeIndexVersion) ------------------ | Branch (659:6): [True: 26, False: 332] ------------------ 660| 26| return -1; 661| | 662| 332| const unsigned char* data = buffer + 1; 663| 332| const unsigned char* data_safe_end = buffer + buffer_size - 4; 664| | 665| 332| unsigned int last[2] = {}; 666| | 667| 20.6k| for (size_t i = 0; i < index_count; ++i) ------------------ | Branch (667:21): [True: 20.3k, False: 262] ------------------ 668| 20.3k| { 669| | // make sure we have enough data to read 670| | // each index reads at most 5 bytes of data; there's a 4 byte tail after data_safe_end 671| | // after this we can be sure we can read without extra bounds checks 672| 20.3k| if (data >= data_safe_end) ------------------ | Branch (672:7): [True: 70, False: 20.3k] ------------------ 673| 70| return -2; 674| | 675| 20.3k| unsigned int v = decodeVByte(data); 676| | 677| | // decode the index of the last baseline 678| 20.3k| unsigned int current = v & 1; 679| 20.3k| v >>= 1; 680| | 681| | // reconstruct index as a delta 682| 20.3k| unsigned int d = (v >> 1) ^ -int(v & 1); 683| 20.3k| unsigned int index = last[current] + d; 684| | 685| | // update last for the next iteration that uses it 686| 20.3k| last[current] = index; 687| | 688| 20.3k| if (index_size == 2) ------------------ | Branch (688:7): [True: 10.1k, False: 10.1k] ------------------ 689| 10.1k| { 690| 10.1k| static_cast(destination)[i] = (unsigned short)(index); 691| 10.1k| } 692| 10.1k| else 693| 10.1k| { 694| 10.1k| static_cast(destination)[i] = index; 695| 10.1k| } 696| 20.3k| } 697| | 698| | // we should've read all data bytes and stopped at the boundary between data and tail 699| 262| if (data != data_safe_end) ------------------ | Branch (699:6): [True: 260, False: 2] ------------------ 700| 260| return -3; 701| | 702| 2| return 0; 703| 262|} indexcodec.cpp:_ZN7meshoptL14pushVertexFifoEPjjRmi: 75| 18.1k|{ 76| 18.1k| fifo[offset] = v; 77| 18.1k| offset = (offset + cond) & 15; 78| 18.1k|} indexcodec.cpp:_ZN7meshoptL12pushEdgeFifoEPA2_jjjRm: 55| 24.4k|{ 56| 24.4k| fifo[offset][0] = a; 57| 24.4k| fifo[offset][1] = b; 58| 24.4k| offset = (offset + 1) & 15; 59| 24.4k|} indexcodec.cpp:_ZN7meshoptL11decodeIndexERPKhj: 125| 4.29k|{ 126| 4.29k| unsigned int v = decodeVByte(data); 127| 4.29k| unsigned int d = (v >> 1) ^ -int(v & 1); 128| | 129| 4.29k| return last + d; 130| 4.29k|} indexcodec.cpp:_ZN7meshoptL13writeTriangleEPvmjjj: 142| 10.2k|{ 143| 10.2k| if (index_size == 2) ------------------ | Branch (143:6): [True: 5.11k, False: 5.11k] ------------------ 144| 5.11k| { 145| 5.11k| unsigned short* tri = static_cast(destination); 146| 5.11k| tri[0] = (unsigned short)(a); 147| 5.11k| tri[1] = (unsigned short)(b); 148| 5.11k| tri[2] = (unsigned short)(c); 149| | 150| 5.11k| return tri + 3; 151| 5.11k| } 152| 5.11k| else 153| 5.11k| { 154| 5.11k| unsigned int* tri = static_cast(destination); 155| 5.11k| tri[0] = a; 156| 5.11k| tri[1] = b; 157| 5.11k| tri[2] = c; 158| | 159| 5.11k| return tri + 3; 160| 5.11k| } 161| 10.2k|} indexcodec.cpp:_ZN7meshoptL11decodeVByteERPKh: 91| 24.6k|{ 92| 24.6k| unsigned char lead = *data++; 93| | 94| | // fast path: single byte 95| 24.6k| if (lead < 128) ------------------ | Branch (95:6): [True: 14.9k, False: 9.63k] ------------------ 96| 14.9k| return lead; 97| | 98| | // slow path: up to 4 extra bytes 99| | // note that this loop always terminates, which is important for malformed data 100| 9.63k| unsigned int result = lead & 127; 101| 9.63k| unsigned int shift = 7; 102| | 103| 27.9k| for (int i = 0; i < 4; ++i) ------------------ | Branch (103:18): [True: 24.4k, False: 3.50k] ------------------ 104| 24.4k| { 105| 24.4k| unsigned char group = *data++; 106| 24.4k| result |= unsigned(group & 127) << shift; 107| 24.4k| shift += 7; 108| | 109| 24.4k| if (group < 128) ------------------ | Branch (109:7): [True: 6.12k, False: 18.2k] ------------------ 110| 6.12k| break; 111| 24.4k| } 112| | 113| 9.63k| return result; 114| 24.6k|} meshopt_encodeMeshletBound: 899| 4.31k|{ 900| 4.31k| size_t codes_size = (max_triangles + 1) / 2; 901| 4.31k| size_t extra_size = max_triangles * 3; 902| | 903| 4.31k| size_t ctrl_size = (max_vertices + 3) / 4; 904| 4.31k| size_t data_size = (max_vertices + 3) / 4 * 16; // worst case: 16 bytes per vertex group 905| | 906| 4.31k| size_t gap_size = (codes_size + ctrl_size < 16) ? 16 - (codes_size + ctrl_size) : 0; ------------------ | Branch (906:20): [True: 2.74k, False: 1.56k] ------------------ 907| | 908| 4.31k| return codes_size + extra_size + ctrl_size + data_size + gap_size; 909| 4.31k|} meshopt_encodeMeshlet: 912| 4.31k|{ 913| 4.31k| using namespace meshopt; 914| | 915| 4.31k| assert(triangle_count <= 256 && vertex_count <= 256); ------------------ | Branch (915:2): [True: 4.31k, False: 0] | Branch (915:2): [True: 4.31k, False: 0] | Branch (915:2): [True: 4.31k, False: 0] ------------------ 916| | 917| | // 4 bits per triangle + up to three bytes of extra data 918| 4.31k| unsigned char codes[256 / 2]; 919| 4.31k| unsigned char extra[256 * 3]; 920| 4.31k| size_t codes_size = (triangle_count + 1) / 2; 921| 4.31k| size_t extra_size = encodeTriangles(codes, extra, triangles, triangle_count); 922| 4.31k| assert(extra_size <= sizeof(extra)); ------------------ | Branch (922:2): [True: 4.31k, False: 0] ------------------ 923| | 924| | // 2 bits per vertex + up to 4 bytes of actual data 925| 4.31k| unsigned char ctrl[256 / 4]; 926| 4.31k| unsigned char data[256 * 4]; 927| 4.31k| size_t ctrl_size = (vertex_count + 3) / 4; 928| 4.31k| size_t data_size = encodeVertices(ctrl, data, vertices, vertex_count); 929| 4.31k| assert(data_size <= sizeof(data)); ------------------ | Branch (929:2): [True: 4.31k, False: 0] ------------------ 930| | 931| | // we need to ensure that up to 16 bytes after extra+data are available for SIMD decoding 932| | // to minimize overhead, we place fixed-size codes+control at the end of the buffer 933| 4.31k| size_t gap_size = (codes_size + ctrl_size < 16) ? 16 - (codes_size + ctrl_size) : 0; ------------------ | Branch (933:20): [True: 2.74k, False: 1.56k] ------------------ 934| | 935| 4.31k| size_t result = codes_size + extra_size + ctrl_size + data_size + gap_size; 936| | 937| 4.31k| if (result > buffer_size) ------------------ | Branch (937:6): [True: 0, False: 4.31k] ------------------ 938| 0| return 0; 939| | 940| | // variable-size data first 941| 4.31k| memcpy(buffer, data, data_size); 942| 4.31k| buffer += data_size; 943| 4.31k| memcpy(buffer, extra, extra_size); 944| 4.31k| buffer += extra_size; 945| | 946| | // gap (for accelerated decoding) separates variable-size and fixed-size data 947| 4.31k| memset(buffer, 0, gap_size); 948| 4.31k| buffer += gap_size; 949| | 950| | // fixed-size data last; it can be located from buffer end during decoding 951| 4.31k| memcpy(buffer, ctrl, ctrl_size); 952| 4.31k| buffer += ctrl_size; 953| 4.31k| memcpy(buffer, codes, codes_size); 954| 4.31k| buffer += codes_size; 955| | 956| |#if TRACE > 1 957| | printf("extra:"); 958| | for (size_t i = 0; i < extra_size; ++i) 959| | printf(" %d", extra[i]); 960| | printf("\n"); 961| | 962| | unsigned int minv = ~0u; 963| | for (size_t i = 0; i < vertex_count; ++i) 964| | minv = minv < vertices[i] ? minv : vertices[i]; 965| | 966| | printf("vertices: [%d+]", minv); 967| | for (size_t i = 0; i < vertex_count; ++i) 968| | printf(" %d", vertices[i] - minv); 969| | printf("\n"); 970| |#endif 971| | 972| |#if TRACE 973| | printf("stats: %d vertices, %d triangles => %d bytes (triangles: %d codes, %d extra; vertices: %d control, %d data; %d gap)\n", 974| | int(vertex_count), int(triangle_count), int(result), 975| | int(codes_size), int(extra_size), int(ctrl_size), int(data_size), int(gap_size)); 976| |#endif 977| | 978| 4.31k| return result; 979| 4.31k|} meshopt_decodeMeshlet: 982| 17.2k|{ 983| 17.2k| using namespace meshopt; 984| | 985| 17.2k| assert(triangle_count <= 256 && vertex_count <= 256); ------------------ | Branch (985:2): [True: 17.2k, False: 0] | Branch (985:2): [True: 17.2k, False: 0] | Branch (985:2): [True: 17.2k, False: 0] ------------------ 986| 17.2k| assert(vertex_size == 4 || vertex_size == 2); ------------------ | Branch (986:2): [True: 10.7k, False: 6.44k] | Branch (986:2): [True: 6.44k, False: 0] | Branch (986:2): [True: 17.2k, False: 0] ------------------ 987| 17.2k| assert(triangle_size == 4 || triangle_size == 3); ------------------ | Branch (987:2): [True: 6.44k, False: 10.7k] | Branch (987:2): [True: 10.7k, False: 0] | Branch (987:2): [True: 17.2k, False: 0] ------------------ 988| | 989| | // layout must match encoding 990| 17.2k| size_t codes_size = (triangle_count + 1) / 2; 991| 17.2k| size_t ctrl_size = (vertex_count + 3) / 4; 992| 17.2k| size_t gap_size = (codes_size + ctrl_size < 16) ? 16 - (codes_size + ctrl_size) : 0; ------------------ | Branch (992:20): [True: 6.85k, False: 10.3k] ------------------ 993| | 994| 17.2k| if (buffer_size < codes_size + ctrl_size + gap_size) ------------------ | Branch (994:6): [True: 4.45k, False: 12.7k] ------------------ 995| 4.45k| return -2; 996| | 997| 12.7k| const unsigned char* end = buffer + buffer_size; 998| 12.7k| const unsigned char* codes = end - codes_size; 999| 12.7k| const unsigned char* ctrl = codes - ctrl_size; 1000| 12.7k| const unsigned char* data = buffer; 1001| | 1002| | // gap ensures we have at least 16 bytes available after bound; this allows SIMD decoders to over-read safely 1003| 12.7k| const unsigned char* bound = ctrl - gap_size; 1004| 12.7k| assert(bound >= buffer && bound + 16 <= buffer + buffer_size); ------------------ | Branch (1004:2): [True: 12.7k, False: 0] | Branch (1004:2): [True: 12.7k, False: 0] | Branch (1004:2): [True: 12.7k, False: 0] ------------------ 1005| | 1006| 12.7k|#if defined(SIMD_FALLBACK) 1007| 12.7k| return (gDecodeTablesInitialized ? decodeMeshletSimd<0> : decodeMeshlet)(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, vertex_size, triangle_size); ------------------ | Branch (1007:10): [True: 12.7k, False: 0] ------------------ 1008| |#elif defined(SIMD_SSE) || defined(SIMD_NEON) 1009| | return decodeMeshletSimd<0>(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, vertex_size, triangle_size); 1010| |#else 1011| | return decodeMeshlet(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, vertex_size, triangle_size); 1012| |#endif 1013| 12.7k|} meshopt_decodeMeshletRaw: 1016| 2.14k|{ 1017| 2.14k| using namespace meshopt; 1018| | 1019| 2.14k| assert(triangle_count <= 256 && vertex_count <= 256); ------------------ | Branch (1019:2): [True: 2.14k, False: 0] | Branch (1019:2): [True: 2.14k, False: 0] | Branch (1019:2): [True: 2.14k, False: 0] ------------------ 1020| | 1021| | // layout must match encoding 1022| 2.14k| size_t codes_size = (triangle_count + 1) / 2; 1023| 2.14k| size_t ctrl_size = (vertex_count + 3) / 4; 1024| 2.14k| size_t gap_size = (codes_size + ctrl_size < 16) ? 16 - (codes_size + ctrl_size) : 0; ------------------ | Branch (1024:20): [True: 340, False: 1.80k] ------------------ 1025| | 1026| 2.14k| if (buffer_size < codes_size + ctrl_size + gap_size) ------------------ | Branch (1026:6): [True: 1.11k, False: 1.02k] ------------------ 1027| 1.11k| return -2; 1028| | 1029| 1.02k| const unsigned char* end = buffer + buffer_size; 1030| 1.02k| const unsigned char* codes = end - codes_size; 1031| 1.02k| const unsigned char* ctrl = codes - ctrl_size; 1032| 1.02k| const unsigned char* data = buffer; 1033| | 1034| | // gap ensures we have at least 16 bytes available after bound; this allows SIMD decoders to over-read safely 1035| 1.02k| const unsigned char* bound = ctrl - gap_size; 1036| 1.02k| assert(bound >= buffer && bound + 16 <= buffer + buffer_size); ------------------ | Branch (1036:2): [True: 1.02k, False: 0] | Branch (1036:2): [True: 1.02k, False: 0] | Branch (1036:2): [True: 1.02k, False: 0] ------------------ 1037| | 1038| 1.02k|#if defined(SIMD_FALLBACK) 1039| 1.02k| return (gDecodeTablesInitialized ? decodeMeshletSimd<1> : decodeMeshlet)(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, 4, 4); ------------------ | Branch (1039:10): [True: 1.02k, False: 0] ------------------ 1040| |#elif defined(SIMD_SSE) || defined(SIMD_NEON) 1041| | return decodeMeshletSimd<1>(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, 4, 4); 1042| |#else 1043| | return decodeMeshlet(vertices, triangles, codes, ctrl, data, bound, vertex_count, triangle_count, 4, 4); 1044| |#endif 1045| 1.02k|} meshletcodec.cpp:_ZN7meshoptL17decodeBuildTablesEv: 398| 2|{ 399| 2|#define NEXT(var, ec) \ 400| 2| shuf[var] = (ec) ? (unsigned char)extra : 15; \ 401| 2| next[var] = (ec) ? 0 : (unsigned char)nextoff; \ 402| 2| extra += (ec), nextoff += 1 - (ec) 403| | 404| | // check for SSE4.1 support if we have a fallback path 405| 2|#if defined(SIMD_SSE) && defined(SIMD_FALLBACK) 406| 2| int cpuinfo[4] = {}; 407| |#ifdef _MSC_VER 408| | __cpuid(cpuinfo, 1); 409| |#else 410| 2| __cpuid(1, cpuinfo[0], cpuinfo[1], cpuinfo[2], cpuinfo[3]); 411| 2|#endif 412| | // bit 19 = SSE4.1 413| 2| if ((cpuinfo[2] & (1 << 19)) == 0) ------------------ | Branch (413:6): [True: 0, False: 2] ------------------ 414| 0| return false; 415| 2|#endif 416| | 417| | // fill triangle decoding tables for each combination of two triangle codes 418| 514| for (int code = 0; code < 256; ++code) ------------------ | Branch (418:21): [True: 512, False: 2] ------------------ 419| 512| { 420| 512| unsigned char shuf[16] = {}; 421| 512| unsigned char next[16] = {}; 422| 512| int extra = 0; 423| 512| int nextoff = 0; 424| | 425| | // state 0..5 will be refilled every iteration, so we ignore that 426| | // state 6..8 will always contain the last decoded triangle because every triangle shifts fifo equally, so we can decode it independently 427| 512| shuf[6] = 12; 428| 512| shuf[7] = 13; 429| 512| shuf[8] = 14; 430| | 431| | // state 15 will contain next (potentially incremented a few times) 432| 512| shuf[15] = 15; 433| | 434| | // state 9..11 will contain the first decoded triangle (tri0), which can refer to extra/next and the original triangle history 435| | // state 12..14 will contain the second decoded triangle (tri1); when decoding edge reuse, we need to handle edge 0/1 specially as it was just decoded earlier 436| 1.53k| for (int k = 0; k < 2; ++k) ------------------ | Branch (436:19): [True: 1.02k, False: 512] ------------------ 437| 1.02k| { 438| 1.02k| int tri = (code >> (k * 4)) & 0xf; 439| | 440| 1.02k| if (tri < 12) ------------------ | Branch (440:8): [True: 768, False: 256] ------------------ 441| 768| { 442| 768| if (k == 1 && tri / 4 == 0) ------------------ | Branch (442:9): [True: 384, False: 384] | Branch (442:19): [True: 128, False: 256] ------------------ 443| 128| { 444| | // we need to decode one of two edges from the triangle we just decoded earlier 445| | // for that we simply need to copy shuf/next values for the two decoded indices 446| 128| shuf[9 + k * 3] = shuf[9 + ((tri & 2) ? 2 : 0)]; ------------------ | Branch (446:34): [True: 64, False: 64] ------------------ 447| 128| next[9 + k * 3] = next[9 + ((tri & 2) ? 2 : 0)]; ------------------ | Branch (447:34): [True: 64, False: 64] ------------------ 448| | 449| 128| shuf[10 + k * 3] = shuf[9 + ((tri & 2) ? 1 : 2)]; ------------------ | Branch (449:35): [True: 64, False: 64] ------------------ 450| 128| next[10 + k * 3] = next[9 + ((tri & 2) ? 1 : 2)]; ------------------ | Branch (450:35): [True: 64, False: 64] ------------------ 451| 128| } 452| 640| else 453| 640| { 454| | // reuse: edge comes from the history based on edge index 455| | // note: we reuse with an offset because last triangle in the original history was consumed by tri0 456| 640| int trioff = 6 + k * 3 + (2 - tri / 4) * 3; 457| | 458| | // edge cb or ac 459| 640| shuf[9 + k * 3] = (unsigned char)(trioff + ((tri & 2) ? 2 : 0)); ------------------ | Branch (459:50): [True: 320, False: 320] ------------------ 460| 640| shuf[10 + k * 3] = (unsigned char)(trioff + ((tri & 2) ? 1 : 2)); ------------------ | Branch (460:51): [True: 320, False: 320] ------------------ 461| 640| } 462| | 463| | // third vertex is either next or comes from extra 464| 768| NEXT(11 + k * 3, tri & 1); ------------------ | | 400| 768| shuf[var] = (ec) ? (unsigned char)extra : 15; \ | | ------------------ | | | Branch (400:14): [True: 384, False: 384] | | ------------------ | | 401| 768| next[var] = (ec) ? 0 : (unsigned char)nextoff; \ | | ------------------ | | | Branch (401:14): [True: 384, False: 384] | | ------------------ | | 402| 768| extra += (ec), nextoff += 1 - (ec) ------------------ 465| 768| } 466| 256| else 467| 256| { 468| | // restart: three vertices, each comes from next or extra 469| 256| int fea = tri > 12; 470| 256| int feb = tri > 13; 471| 256| int fec = tri > 14; 472| | 473| 256| NEXT(9 + k * 3, fea); ------------------ | | 400| 256| shuf[var] = (ec) ? (unsigned char)extra : 15; \ | | ------------------ | | | Branch (400:14): [True: 192, False: 64] | | ------------------ | | 401| 256| next[var] = (ec) ? 0 : (unsigned char)nextoff; \ | | ------------------ | | | Branch (401:14): [True: 192, False: 64] | | ------------------ | | 402| 256| extra += (ec), nextoff += 1 - (ec) ------------------ 474| 256| NEXT(10 + k * 3, feb); ------------------ | | 400| 256| shuf[var] = (ec) ? (unsigned char)extra : 15; \ | | ------------------ | | | Branch (400:14): [True: 128, False: 128] | | ------------------ | | 401| 256| next[var] = (ec) ? 0 : (unsigned char)nextoff; \ | | ------------------ | | | Branch (401:14): [True: 128, False: 128] | | ------------------ | | 402| 256| extra += (ec), nextoff += 1 - (ec) ------------------ 475| 256| NEXT(11 + k * 3, fec); ------------------ | | 400| 256| shuf[var] = (ec) ? (unsigned char)extra : 15; \ | | ------------------ | | | Branch (400:14): [True: 64, False: 192] | | ------------------ | | 401| 256| next[var] = (ec) ? 0 : (unsigned char)nextoff; \ | | ------------------ | | | Branch (401:14): [True: 64, False: 192] | | ------------------ | | 402| 256| extra += (ec), nextoff += 1 - (ec) ------------------ 476| 256| } 477| 1.02k| } 478| | 479| | // next needs to advance 480| 512| next[15] = (unsigned char)nextoff; 481| | 482| | // next[0..8] = 0 trivially (never written to); next[9] must also be 0 because nextoff is 0 initially 483| | // shuf[0..5] is not used, which allows us to pack next[10..15] + shuf[6..15] into a single 16-byte entry 484| 512| assert(next[9] == 0); ------------------ | Branch (484:3): [True: 512, False: 0] ------------------ 485| 512| memcpy(&kDecodeTableMasks[code][0], &next[10], 6); 486| 512| memcpy(&kDecodeTableMasks[code][6], &shuf[6], 10); 487| 512| kDecodeTableExtra[code] = (unsigned char)extra; 488| 512| } 489| | 490| | // fill vertex decoding tables for each combination of four vertex references 491| 514| for (unsigned int i = 0; i < 256; ++i) ------------------ | Branch (491:27): [True: 512, False: 2] ------------------ 492| 512| { 493| 512| unsigned char shuf[16] = {}; 494| 512| int offset = 0; 495| | 496| 2.56k| for (int k = 0; k < 4; ++k) ------------------ | Branch (496:19): [True: 2.04k, False: 512] ------------------ 497| 2.04k| { 498| 2.04k| int code = ((i >> k) & 1) | ((i >> (k + 3)) & 2); 499| 2.04k| int length = i == 0xff ? 4 : code; // 0/1/2/3 bytes, or all 4 bytes if code==0xff ------------------ | Branch (499:17): [True: 8, False: 2.04k] ------------------ 500| | 501| 2.04k| shuf[k * 4 + 0] = (length > 0) ? (unsigned char)(offset + 0) : 0x80; ------------------ | Branch (501:22): [True: 1.53k, False: 512] ------------------ 502| 2.04k| shuf[k * 4 + 1] = (length > 1) ? (unsigned char)(offset + 1) : 0x80; ------------------ | Branch (502:22): [True: 1.02k, False: 1.02k] ------------------ 503| 2.04k| shuf[k * 4 + 2] = (length > 2) ? (unsigned char)(offset + 2) : 0x80; ------------------ | Branch (503:22): [True: 512, False: 1.53k] ------------------ 504| 2.04k| shuf[k * 4 + 3] = (length > 3) ? (unsigned char)(offset + 3) : 0x80; ------------------ | Branch (504:22): [True: 8, False: 2.04k] ------------------ 505| | 506| 2.04k| offset += length; 507| 2.04k| } 508| | 509| 512| memcpy(kDecodeTableVerts[i], shuf, sizeof(shuf)); 510| 512| kDecodeTableLength[i] = (unsigned char)offset; 511| 512| } 512| | 513| 2| return true; 514| | 515| 2|#undef NEXT 516| 2|} meshletcodec.cpp:_ZN7meshoptL15encodeTrianglesEPhS0_PKhm: 109| 4.31k|{ 110| 4.31k| EdgeFifo8 edgefifo; 111| 4.31k| memset(edgefifo, -1, sizeof(edgefifo)); 112| | 113| 4.31k| size_t edgefifooffset = 0; 114| | 115| 4.31k| unsigned int next = 0; 116| | 117| | // 4-bit triangle codes give us 16 options that we use as follows: 118| | // 3*2 edge reuse (2 edges * 3 last triangles) * 2 next/explicit = 12 options 119| | // 4 remaining options = next bits; 000, 001, 011, 111. 120| | // triangles are rotated to make next bits line up. 121| 4.31k| memset(codes, 0, (triangle_count + 1) / 2); 122| | 123| 4.31k| static const int rotations[] = {0, 1, 2, 0, 1}; 124| | 125| 4.31k| unsigned char* start = extra; 126| | 127| 184k| for (size_t i = 0; i < triangle_count; ++i) ------------------ | Branch (127:21): [True: 180k, False: 4.31k] ------------------ 128| 180k| { 129| |#if TRACE > 1 130| | unsigned int last = next; 131| |#endif 132| | 133| 180k| int fer = getEdgeFifo8(edgefifo, triangles[i * 3 + 0], triangles[i * 3 + 1], triangles[i * 3 + 2], edgefifooffset); 134| | 135| 180k| if (fer >= 0 && (fer >> 2) < 6) ------------------ | Branch (135:7): [True: 88.2k, False: 91.8k] | Branch (135:19): [True: 84.4k, False: 3.82k] ------------------ 136| 84.4k| { 137| | // note: getEdgeFifo8 implicitly rotates triangles by matching a/b to existing edge 138| 84.4k| const int* order = rotations + (fer & 3); 139| | 140| 84.4k| unsigned int a = triangles[i * 3 + order[0]], b = triangles[i * 3 + order[1]], c = triangles[i * 3 + order[2]]; 141| | 142| 84.4k| int fec = (c == next) ? (next++, 0) : 1; ------------------ | Branch (142:14): [True: 767, False: 83.6k] ------------------ 143| | 144| |#if TRACE > 1 145| | printf("%3d+ | %3d %3d %3d | edge: e%d c%d\n", last, a, b, c, fer >> 2, fec); 146| |#endif 147| | 148| 84.4k| unsigned int code = (fer >> 2) * 2 + fec; 149| | 150| 84.4k| codes[i / 2] |= (unsigned char)(code << ((i & 1) * 4)); 151| | 152| 84.4k| if (fec) ------------------ | Branch (152:8): [True: 83.6k, False: 767] ------------------ 153| 83.6k| *extra++ = (unsigned char)c; 154| | 155| 84.4k| pushEdgeFifo8(edgefifo, c, b, edgefifooffset); 156| 84.4k| pushEdgeFifo8(edgefifo, a, c, edgefifooffset); 157| 84.4k| } 158| 95.6k| else 159| 95.6k| { 160| | // rotate triangles to minimize the need for extra vertices 161| 95.6k| int rotation = rotateTriangle(triangles[i * 3 + 0], triangles[i * 3 + 1], triangles[i * 3 + 2]); 162| 95.6k| const int* order = rotations + rotation; 163| | 164| 95.6k| unsigned int a = triangles[i * 3 + order[0]], b = triangles[i * 3 + order[1]], c = triangles[i * 3 + order[2]]; 165| | 166| | // fe must be continuous: once a vertex is encoded with next, further vertices must also be encoded with next 167| 95.6k| int fea = (a == next && b == next + 1 && c == next + 2) ? (next++, 0) : 1; ------------------ | Branch (167:15): [True: 8.32k, False: 87.3k] | Branch (167:28): [True: 3.68k, False: 4.64k] | Branch (167:45): [True: 3.25k, False: 423] ------------------ 168| 95.6k| int feb = (b == next && c == next + 1) ? (next++, 0) : 1; ------------------ | Branch (168:15): [True: 8.71k, False: 86.9k] | Branch (168:28): [True: 3.49k, False: 5.22k] ------------------ 169| 95.6k| int fec = (c == next) ? (next++, 0) : 1; ------------------ | Branch (169:14): [True: 4.54k, False: 91.1k] ------------------ 170| | 171| 95.6k| assert(fea == 1 || feb == 0); ------------------ | Branch (171:4): [True: 92.3k, False: 3.25k] | Branch (171:4): [True: 3.25k, False: 0] | Branch (171:4): [True: 95.6k, False: 0] ------------------ 172| 95.6k| assert(feb == 1 || fec == 0); ------------------ | Branch (172:4): [True: 92.1k, False: 3.49k] | Branch (172:4): [True: 3.49k, False: 0] | Branch (172:4): [True: 95.6k, False: 0] ------------------ 173| | 174| |#if TRACE > 1 175| | printf("%3d+ | %3d %3d %3d | restart: %d%d%d\n", last, a, b, c, fea, feb, fec); 176| |#endif 177| | 178| 95.6k| unsigned int code = 12 + (fea + feb + fec); 179| | 180| 95.6k| codes[i / 2] |= (unsigned char)(code << ((i & 1) * 4)); 181| | 182| 95.6k| if (fea) ------------------ | Branch (182:8): [True: 92.3k, False: 3.25k] ------------------ 183| 92.3k| *extra++ = (unsigned char)a; 184| 95.6k| if (feb) ------------------ | Branch (184:8): [True: 92.1k, False: 3.49k] ------------------ 185| 92.1k| *extra++ = (unsigned char)b; 186| 95.6k| if (fec) ------------------ | Branch (186:8): [True: 91.1k, False: 4.54k] ------------------ 187| 91.1k| *extra++ = (unsigned char)c; 188| | 189| 95.6k| pushEdgeFifo8(edgefifo, c, b, edgefifooffset); 190| 95.6k| pushEdgeFifo8(edgefifo, a, c, edgefifooffset); 191| 95.6k| } 192| 180k| } 193| | 194| 4.31k| return extra - start; 195| 4.31k|} meshletcodec.cpp:_ZN7meshoptL12getEdgeFifo8EPA2_jjjjm: 82| 180k|{ 83| 979k| for (int i = 0; i < 8; ++i) ------------------ | Branch (83:18): [True: 887k, False: 91.8k] ------------------ 84| 887k| { 85| 887k| size_t index = (offset - 1 - i) & 7; 86| | 87| 887k| unsigned int e0 = fifo[index][0]; 88| 887k| unsigned int e1 = fifo[index][1]; 89| | 90| 887k| if (e0 == a && e1 == b) ------------------ | Branch (90:7): [True: 138k, False: 749k] | Branch (90:18): [True: 71.1k, False: 67.6k] ------------------ 91| 71.1k| return (i << 2) | 0; 92| 816k| if (e0 == b && e1 == c) ------------------ | Branch (92:7): [True: 55.7k, False: 761k] | Branch (92:18): [True: 9.11k, False: 46.6k] ------------------ 93| 9.11k| return (i << 2) | 1; 94| 807k| if (e0 == c && e1 == a) ------------------ | Branch (94:7): [True: 56.9k, False: 750k] | Branch (94:18): [True: 7.99k, False: 48.9k] ------------------ 95| 7.99k| return (i << 2) | 2; 96| 807k| } 97| | 98| 91.8k| return -1; 99| 180k|} meshletcodec.cpp:_ZN7meshoptL13pushEdgeFifo8EPA2_jjjRm: 102| 360k|{ 103| 360k| fifo[offset][0] = a; 104| 360k| fifo[offset][1] = b; 105| 360k| offset = (offset + 1) & 7; 106| 360k|} meshletcodec.cpp:_ZN7meshoptL14rotateTriangleEjjj: 77| 95.6k|{ 78| 95.6k| return (a > b && a > c) ? 1 : (b > c ? 2 : 0); ------------------ | Branch (78:10): [True: 40.3k, False: 55.2k] | Branch (78:19): [True: 27.4k, False: 12.9k] | Branch (78:33): [True: 26.9k, False: 41.2k] ------------------ 79| 95.6k|} meshletcodec.cpp:_ZN7meshoptL14encodeVerticesEPhS0_PKjm: 198| 4.31k|{ 199| | // grouped varint, 2 bit per value to indicate 0/1/2/3 byte deltas, with per-group 4-byte fallback 200| 4.31k| memset(ctrl, 0, (vertex_count + 3) / 4); 201| | 202| 4.31k| unsigned char* start = data; 203| | 204| 4.31k| unsigned int last = ~0u; 205| | 206| 46.4k| for (size_t i = 0; i < vertex_count; i += 4) ------------------ | Branch (206:21): [True: 42.1k, False: 4.31k] ------------------ 207| 42.1k| { 208| 42.1k| unsigned int gv[4] = {}; 209| | 210| 208k| for (int k = 0; k < 4 && i + k < vertex_count; ++k) ------------------ | Branch (210:19): [True: 167k, False: 41.0k] | Branch (210:28): [True: 166k, False: 1.09k] ------------------ 211| 166k| { 212| 166k| unsigned int d = vertices[i + k] - last - 1; 213| 166k| unsigned int v = (d << 1) ^ (int(d) >> 31); 214| | 215| 166k| gv[k] = v; 216| 166k| last = vertices[i + k]; 217| 166k| } 218| | 219| | // if any value needs 4 bytes, or if *all* values need 3 bytes, we use 4 bytes for all values 220| | // this allows us to encode most 3-byte deltas with 3 bytes which saves space overall 221| 42.1k| bool use4 = (gv[0] | gv[1] | gv[2] | gv[3]) > 0xffffff || (gv[0] > 0xffff && gv[1] > 0xffff && gv[2] > 0xffff && gv[3] > 0xffff); ------------------ | Branch (221:15): [True: 29.0k, False: 13.0k] | Branch (221:62): [True: 1.29k, False: 11.7k] | Branch (221:80): [True: 799, False: 499] | Branch (221:98): [True: 458, False: 341] | Branch (221:116): [True: 348, False: 110] ------------------ 222| | 223| 210k| for (int k = 0; k < 4; ++k) ------------------ | Branch (223:19): [True: 168k, False: 42.1k] ------------------ 224| 168k| { 225| 168k| unsigned int v = gv[k]; 226| | 227| | // 0/1/2/3 bytes per value, or all 4 values use 4 bytes 228| 168k| int code = use4 ? 3 : (v == 0 ? 0 : (v < 256 ? 1 : (v < 65536 ? 2 : 3))); ------------------ | Branch (228:15): [True: 117k, False: 50.8k] | Branch (228:27): [True: 2.02k, False: 48.8k] | Branch (228:41): [True: 43.2k, False: 5.58k] | Branch (228:56): [True: 2.78k, False: 2.80k] ------------------ 229| | 230| 168k| if (code > 0) ------------------ | Branch (230:8): [True: 166k, False: 2.02k] ------------------ 231| 166k| *data++ = (unsigned char)(v & 0xff); 232| 168k| if (code > 1) ------------------ | Branch (232:8): [True: 123k, False: 45.2k] ------------------ 233| 123k| *data++ = (unsigned char)((v >> 8) & 0xff); 234| 168k| if (code > 2) ------------------ | Branch (234:8): [True: 120k, False: 48.0k] ------------------ 235| 120k| *data++ = (unsigned char)((v >> 16) & 0xff); 236| 168k| if (use4) ------------------ | Branch (236:8): [True: 117k, False: 50.8k] ------------------ 237| 117k| *data++ = (unsigned char)((v >> 24) & 0xff); 238| | 239| | // split low and high bits into two nibbles for better packing 240| 168k| ctrl[i / 4] |= ((code & 1) << k) | ((code >> 1) << (k + 4)); 241| 168k| } 242| 42.1k| } 243| | 244| 4.31k| return data - start; 245| 4.31k|} meshletcodec.cpp:_ZN7meshoptL17decodeMeshletSimdILi0EEEiPvS1_PKhS3_S3_S3_mmmm: 865| 12.7k|{ 866| 12.7k| assert(gDecodeTablesInitialized); ------------------ | Branch (866:2): [True: 12.7k, False: 0] ------------------ 867| 12.7k| (void)gDecodeTablesInitialized; 868| | 869| 12.7k|#ifdef __clang__ 870| | // data is guaranteed to be non-null initially; if decode loops never hit bounds errors, it remains non-null 871| 12.7k| __builtin_assume(data); 872| 12.7k|#endif 873| | 874| | // decodes 4 vertices at a time with tail processing; writes up to align(vertex_size * vertex_count, 4) 875| | // raw decoding skips tail processing by rounding up vertex count; it's safe because output buffer is guaranteed to have extra space, and tail control data is 0 876| 12.7k| if (vertex_size == 4 || Raw) ------------------ | Branch (876:6): [True: 8.52k, False: 4.21k] | Branch (876:26): [Folded, False: 0] ------------------ 877| 8.52k| data = decodeVerticesSimd(static_cast(vertices), ctrl, data, bound, Raw ? (vertex_count + 3) & ~3 : vertex_count); ------------------ | Branch (877:86): [Folded, False: 8.52k] ------------------ 878| 4.21k| else 879| 4.21k| data = decodeVerticesSimd(static_cast(vertices), ctrl, data, bound, vertex_count); 880| 12.7k| if (!data) ------------------ | Branch (880:6): [True: 852, False: 11.8k] ------------------ 881| 852| return -2; 882| | 883| | // decodes 2/4 triangles at a time with tail processing; writes up to align(triangle_size * triangle_count, 4) 884| | // raw decoding skips tail processing by rounding up triangle count; it's safe because output buffer is guaranteed to have extra space, and tail code data is 0 885| 11.8k| if (triangle_size == 4 || Raw) ------------------ | Branch (885:6): [True: 3.78k, False: 8.10k] | Branch (885:28): [Folded, False: 0] ------------------ 886| 3.78k| data = decodeTrianglesSimd(static_cast(triangles), codes, data, bound, Raw ? (triangle_count + 1) & ~1 : triangle_count); ------------------ | Branch (886:89): [Folded, False: 3.78k] ------------------ 887| 8.10k| else 888| 8.10k| data = decodeTrianglesSimd(static_cast(triangles), codes, data, bound, triangle_count); 889| 11.8k| if (!data) ------------------ | Branch (889:6): [True: 226, False: 11.6k] ------------------ 890| 226| return -2; 891| | 892| 11.6k| return (data == bound) ? 0 : -3; ------------------ | Branch (892:9): [True: 8.68k, False: 2.97k] ------------------ 893| 11.8k|} meshletcodec.cpp:_ZN7meshoptL18decodeVerticesSimdEPjPKhS2_S2_m: 750| 9.55k|{ 751| 9.55k|#if defined(SIMD_SSE) 752| 9.55k| __m128i last = _mm_set1_epi32(-1); 753| |#elif defined(SIMD_NEON) 754| | uint32x4_t last = vdupq_n_u32(~0u); 755| |#endif 756| | 757| 9.55k| size_t groups = vertex_count / 4; 758| | 759| | // process all complete groups 760| 131k| for (size_t i = 0; i < groups; ++i) ------------------ | Branch (760:21): [True: 122k, False: 8.93k] ------------------ 761| 122k| { 762| 122k| unsigned char code = *ctrl++; 763| 122k| if (data > bound) ------------------ | Branch (763:7): [True: 627, False: 122k] ------------------ 764| 627| return NULL; 765| | 766| 122k| last = decodeVertexGroup(last, code, data); 767| | 768| 122k|#if defined(SIMD_SSE) 769| 122k| _mm_storeu_si128(reinterpret_cast<__m128i*>(&vertices[i * 4]), last); 770| |#elif defined(SIMD_NEON) 771| | vst1q_u32(&vertices[i * 4], last); 772| |#endif 773| 122k| } 774| | 775| | // process a 1-3 vertex tail; to maintain the memory safety guarantee we have to write individual elements 776| 8.93k| if (vertex_count & 3) ------------------ | Branch (776:6): [True: 2.53k, False: 6.40k] ------------------ 777| 2.53k| { 778| 2.53k| unsigned char code = *ctrl++; 779| | 780| 2.53k| if (data > bound) ------------------ | Branch (780:7): [True: 12, False: 2.51k] ------------------ 781| 12| return NULL; 782| | 783| 2.51k| last = decodeVertexGroup(last, code, data); 784| | 785| 2.51k| unsigned int* tail = &vertices[vertex_count & ~3u]; 786| | 787| 2.51k|#if defined(SIMD_SSE) 788| 2.51k| tail[0] = _mm_cvtsi128_si32(last); 789| 2.51k| if ((vertex_count & 3) > 1) ------------------ | Branch (789:7): [True: 1.14k, False: 1.37k] ------------------ 790| 1.14k| tail[1] = _mm_extract_epi32(last, 1); 791| 2.51k| if ((vertex_count & 3) > 2) ------------------ | Branch (791:7): [True: 422, False: 2.09k] ------------------ 792| 422| tail[2] = _mm_extract_epi32(last, 2); 793| |#elif defined(SIMD_NEON) 794| | vst1q_lane_u32(&tail[0], last, 0); 795| | if ((vertex_count & 3) > 1) 796| | vst1q_lane_u32(&tail[1], last, 1); 797| | if ((vertex_count & 3) > 2) 798| | vst1q_lane_u32(&tail[2], last, 2); 799| |#endif 800| 2.51k| } 801| | 802| 8.91k| return data; 803| 8.93k|} _ZN7meshopt17decodeVertexGroupEDv2_xhRPKh: 540| 221k|{ 541| 221k| __m128i word = _mm_loadu_si128(reinterpret_cast(data)); 542| 221k| __m128i shuf = _mm_loadu_si128(reinterpret_cast(kDecodeTableVerts[code])); 543| | 544| 221k| __m128i v = _mm_shuffle_epi8(word, shuf); 545| | 546| | // unzigzag+1 547| 221k| __m128i xl = _mm_sub_epi32(_mm_setzero_si128(), _mm_and_si128(v, _mm_set1_epi32(1))); 548| 221k| __m128i xr = _mm_srli_epi32(v, 1); 549| 221k| __m128i x = _mm_add_epi32(_mm_xor_si128(xl, xr), _mm_set1_epi32(1)); 550| | 551| | // prefix sum 552| 221k| x = _mm_add_epi32(x, _mm_slli_si128(x, 8)); 553| 221k| x = _mm_add_epi32(x, _mm_slli_si128(x, 4)); 554| 221k| x = _mm_add_epi32(x, _mm_shuffle_epi32(last, 0xff)); 555| | 556| 221k| data += kDecodeTableLength[code]; 557| | 558| 221k| return x; 559| 221k|} meshletcodec.cpp:_ZN7meshoptL18decodeVerticesSimdEPtPKhS2_S2_m: 807| 4.21k|{ 808| 4.21k|#if defined(SIMD_SSE) 809| 4.21k| __m128i repack = _mm_setr_epi8(0, 1, 4, 5, 8, 9, 12, 13, 0, 0, 0, 0, 0, 0, 0, 0); 810| 4.21k| __m128i last = _mm_set1_epi32(-1); 811| |#elif defined(SIMD_NEON) 812| | uint32x4_t last = vdupq_n_u32(~0u); 813| |#endif 814| | 815| | // because the output buffer is guaranteed to have 32-bit aligned size available, we can simplify tail processing 816| | // if the number of vertices mod 4 is 3, we'd normally need to write 8+6 bytes, but we can instead overwrite up to 2 bytes in the main loop 817| 4.21k| size_t groups = (vertex_count + 1) / 4; 818| | 819| | // process all complete groups 820| 99.2k| for (size_t i = 0; i < groups; ++i) ------------------ | Branch (820:21): [True: 95.4k, False: 3.79k] ------------------ 821| 95.4k| { 822| 95.4k| unsigned char code = *ctrl++; 823| | 824| 95.4k| if (data > bound) ------------------ | Branch (824:7): [True: 416, False: 95.0k] ------------------ 825| 416| return NULL; 826| | 827| 95.0k| last = decodeVertexGroup(last, code, data); 828| | 829| 95.0k|#if defined(SIMD_SSE) 830| 95.0k| __m128i r = _mm_shuffle_epi8(last, repack); 831| 95.0k| _mm_storel_epi64(reinterpret_cast<__m128i*>(&vertices[i * 4]), r); 832| |#elif defined(SIMD_NEON) 833| | uint16x4_t r = vmovn_u32(last); 834| | vst1_u16(&vertices[i * 4], r); 835| |#endif 836| 95.0k| } 837| | 838| | // process a 1-2 vertex tail; to maintain the memory safety guarantee we have to write a 32-bit element 839| 3.79k| if (groups * 4 < vertex_count) ------------------ | Branch (839:6): [True: 2.10k, False: 1.69k] ------------------ 840| 2.10k| { 841| 2.10k| unsigned char code = *ctrl++; 842| | 843| 2.10k| if (data > bound) ------------------ | Branch (843:7): [True: 10, False: 2.09k] ------------------ 844| 10| return NULL; 845| | 846| 2.09k| last = decodeVertexGroup(last, code, data); 847| | 848| 2.09k| unsigned short* tail = &vertices[vertex_count & ~3u]; 849| | 850| 2.09k|#if defined(SIMD_SSE) 851| 2.09k| __m128i r = _mm_shufflelo_epi16(last, 8); 852| 2.09k| *reinterpret_cast(tail) = _mm_cvtsi128_si32(r); 853| |#elif defined(SIMD_NEON) 854| | uint16x4_t r = vmovn_u32(last); 855| | vst1_lane_u32(reinterpret_cast(tail), vreinterpret_u32_u16(r), 0); 856| |#endif 857| 2.09k| } 858| | 859| 3.78k| return data; 860| 3.79k|} meshletcodec.cpp:_ZN7meshoptL19decodeTrianglesSimdEPjPKhS2_S2_m: 615| 4.60k|{ 616| 4.60k|#if defined(SIMD_SSE) 617| 4.60k| __m128i repack = _mm_setr_epi8(9, 10, 11, -1, 12, 13, 14, -1, 0, 0, 0, 0, 0, 0, 0, 0); 618| 4.60k| __m128i state = _mm_setzero_si128(); 619| |#elif defined(SIMD_NEON) 620| | uint8x8_t repack = vcreate_u8(0xff0e0d0cff0b0a09ull); 621| | uint8x16_t state = vdupq_n_u8(0); 622| |#endif 623| | 624| 4.60k| size_t groups = triangle_count / 2; 625| | 626| | // process all complete groups 627| 205k| for (size_t i = 0; i < groups; ++i) ------------------ | Branch (627:21): [True: 200k, False: 4.46k] ------------------ 628| 200k| { 629| 200k| unsigned char code = *codes++; 630| | 631| 200k| if (extra > bound) ------------------ | Branch (631:7): [True: 138, False: 200k] ------------------ 632| 138| return NULL; 633| | 634| 200k| state = decodeTriangleGroup(state, code, extra); 635| | 636| | // write 6 bytes of new triangle data into output, formatted as 8 bytes with 0 padding 637| 200k|#if defined(SIMD_SSE) 638| 200k| __m128i r = _mm_shuffle_epi8(state, repack); 639| 200k| _mm_storel_epi64(reinterpret_cast<__m128i*>(&triangles[i * 2]), r); 640| |#elif defined(SIMD_NEON) 641| | uint32x2_t r = vreinterpret_u32_u8(vqtbl1_u8(state, repack)); 642| | vst1_u32(&triangles[i * 2], r); 643| |#endif 644| 200k| } 645| | 646| | // process a 1 triangle tail; to maintain the memory safety guarantee we have to write a 32-bit element 647| 4.46k| if (triangle_count & 1) ------------------ | Branch (647:6): [True: 1.70k, False: 2.76k] ------------------ 648| 1.70k| { 649| 1.70k| unsigned char code = *codes++; 650| | 651| 1.70k| if (extra > bound) ------------------ | Branch (651:7): [True: 42, False: 1.65k] ------------------ 652| 42| return NULL; 653| | 654| 1.65k| state = decodeTriangleGroup(state, code, extra); 655| | 656| 1.65k| unsigned int* tail = &triangles[triangle_count & ~1u]; 657| | 658| 1.65k|#if defined(SIMD_SSE) 659| 1.65k| __m128i r = _mm_shuffle_epi8(state, repack); 660| 1.65k| *tail = unsigned(_mm_cvtsi128_si32(r)); 661| |#elif defined(SIMD_NEON) 662| | uint32x2_t r = vreinterpret_u32_u8(vqtbl1_u8(state, repack)); 663| | vst1_lane_u32(tail, r, 0); 664| |#endif 665| 1.65k| } 666| | 667| 4.42k| return extra; 668| 4.46k|} _ZN7meshopt19decodeTriangleGroupEDv2_xhRPKh: 524| 371k|{ 525| 371k| __m128i shuf = _mm_loadu_si128(reinterpret_cast(kDecodeTableMasks[code])); 526| 371k| __m128i next = _mm_slli_si128(shuf, 10); 527| | 528| | // patch first 6 bytes with current extra and roll state forward 529| 371k| __m128i ext = _mm_loadl_epi64(reinterpret_cast(extra)); 530| 371k| state = _mm_blend_epi16(state, ext, 7); 531| 371k| state = _mm_add_epi8(_mm_shuffle_epi8(state, shuf), next); 532| | 533| 371k| extra += kDecodeTableExtra[code]; 534| | 535| 371k| return state; 536| 371k|} meshletcodec.cpp:_ZN7meshoptL19decodeTrianglesSimdEPhPKhS2_S2_m: 672| 8.10k|{ 673| 8.10k|#if defined(SIMD_SSE) 674| 8.10k| __m128i state = _mm_setzero_si128(); 675| |#elif defined(SIMD_NEON) 676| | uint8x16_t state = vdupq_n_u8(0); 677| |#endif 678| | 679| | // because the output buffer is guaranteed to have 32-bit aligned size available, we can optimize writes and tail processing 680| | // instead of processing triangles 2 at a time, we process 2 *pairs* at a time (12-byte write) followed by a tail pair, if present 681| | // if the number of triangles mod 4 is 3, we'd normally need to write 12k+9 bytes, but we can instead overwrite up to 3 bytes in the main loop 682| 8.10k| size_t groups = (triangle_count + 1) / 4; 683| | 684| | // process all complete groups 685| 89.5k| for (size_t i = 0; i < groups; ++i) ------------------ | Branch (685:21): [True: 81.4k, False: 8.03k] ------------------ 686| 81.4k| { 687| 81.4k| unsigned char code0 = *codes++; 688| 81.4k| unsigned char code1 = *codes++; 689| | 690| | // each triangle pair reads <=6 bytes from extra, so two pairs need <=12 bytes and gap guarantees 16 byte of overread 691| 81.4k| if (extra > bound) ------------------ | Branch (691:7): [True: 72, False: 81.4k] ------------------ 692| 72| return NULL; 693| | 694| 81.4k| state = decodeTriangleGroup(state, code0, extra); 695| | 696| | // write first decoded triangle and first index of second decoded triangle 697| 81.4k|#if defined(SIMD_SSE) 698| 81.4k| __m128i r0 = _mm_srli_si128(state, 9); 699| 81.4k| *reinterpret_cast(&triangles[i * 12]) = _mm_cvtsi128_si32(r0); 700| |#elif defined(SIMD_NEON) 701| | uint8x16_t r0 = vextq_u8(state, vdupq_n_u8(0), 9); 702| | vst1q_lane_u32(reinterpret_cast(&triangles[i * 12]), vreinterpretq_u32_u8(r0), 0); 703| |#endif 704| | 705| 81.4k| state = decodeTriangleGroup(state, code1, extra); 706| | 707| | // write last two indices of second decoded triangle that we didn't write above plus two new ones 708| | // note that the second decoded triangle has shifted down to 6-8 bytes, hence shift by 7 709| 81.4k|#if defined(SIMD_SSE) 710| 81.4k| __m128i r1 = _mm_srli_si128(state, 7); 711| 81.4k| _mm_storel_epi64(reinterpret_cast<__m128i*>(&triangles[i * 12 + 4]), r1); 712| |#elif defined(SIMD_NEON) 713| | uint8x16_t r1 = vextq_u8(state, vdupq_n_u8(0), 7); 714| | vst1_u8(&triangles[i * 12 + 4], vget_low_u8(r1)); 715| |#endif 716| 81.4k| } 717| | 718| | // process a 1-2 triangle tail; to maintain the memory safety guarantee we have to write 1-2 32-bit elements 719| 8.03k| if (groups * 4 < triangle_count) ------------------ | Branch (719:6): [True: 6.32k, False: 1.70k] ------------------ 720| 6.32k| { 721| 6.32k| unsigned char code = *codes++; 722| | 723| 6.32k| if (extra > bound) ------------------ | Branch (723:7): [True: 34, False: 6.29k] ------------------ 724| 34| return NULL; 725| | 726| 6.29k| state = decodeTriangleGroup(state, code, extra); 727| | 728| 6.29k| unsigned char* tail = &triangles[(triangle_count & ~3u) * 3]; 729| | 730| 6.29k|#if defined(SIMD_SSE) 731| 6.29k| __m128i r = _mm_srli_si128(state, 9); 732| | 733| 6.29k| *reinterpret_cast(tail) = _mm_cvtsi128_si32(r); 734| 6.29k| if ((triangle_count & 3) > 1) ------------------ | Branch (734:7): [True: 738, False: 5.55k] ------------------ 735| 738| *reinterpret_cast(tail + 4) = _mm_extract_epi32(r, 1); 736| |#elif defined(SIMD_NEON) 737| | uint8x16_t r = vextq_u8(state, vdupq_n_u8(0), 9); 738| | 739| | vst1q_lane_u32(reinterpret_cast(tail), vreinterpretq_u32_u8(r), 0); 740| | if ((triangle_count & 3) > 1) 741| | vst1q_lane_u32(reinterpret_cast(tail + 4), vreinterpretq_u32_u8(r), 1); 742| |#endif 743| 6.29k| } 744| | 745| 7.99k| return extra; 746| 8.03k|} meshletcodec.cpp:_ZN7meshoptL17decodeMeshletSimdILi1EEEiPvS1_PKhS3_S3_S3_mmmm: 865| 1.02k|{ 866| 1.02k| assert(gDecodeTablesInitialized); ------------------ | Branch (866:2): [True: 1.02k, False: 0] ------------------ 867| 1.02k| (void)gDecodeTablesInitialized; 868| | 869| 1.02k|#ifdef __clang__ 870| | // data is guaranteed to be non-null initially; if decode loops never hit bounds errors, it remains non-null 871| 1.02k| __builtin_assume(data); 872| 1.02k|#endif 873| | 874| | // decodes 4 vertices at a time with tail processing; writes up to align(vertex_size * vertex_count, 4) 875| | // raw decoding skips tail processing by rounding up vertex count; it's safe because output buffer is guaranteed to have extra space, and tail control data is 0 876| 1.02k| if (vertex_size == 4 || Raw) ------------------ | Branch (876:6): [True: 1.02k, False: 0] | Branch (876:26): [True: 0, Folded] ------------------ 877| 1.02k| data = decodeVerticesSimd(static_cast(vertices), ctrl, data, bound, Raw ? (vertex_count + 3) & ~3 : vertex_count); ------------------ | Branch (877:86): [True: 1.02k, Folded] ------------------ 878| 0| else 879| 0| data = decodeVerticesSimd(static_cast(vertices), ctrl, data, bound, vertex_count); 880| 1.02k| if (!data) ------------------ | Branch (880:6): [True: 213, False: 816] ------------------ 881| 213| return -2; 882| | 883| | // decodes 2/4 triangles at a time with tail processing; writes up to align(triangle_size * triangle_count, 4) 884| | // raw decoding skips tail processing by rounding up triangle count; it's safe because output buffer is guaranteed to have extra space, and tail code data is 0 885| 816| if (triangle_size == 4 || Raw) ------------------ | Branch (885:6): [True: 816, False: 0] | Branch (885:28): [True: 0, Folded] ------------------ 886| 816| data = decodeTrianglesSimd(static_cast(triangles), codes, data, bound, Raw ? (triangle_count + 1) & ~1 : triangle_count); ------------------ | Branch (886:89): [True: 816, Folded] ------------------ 887| 0| else 888| 0| data = decodeTrianglesSimd(static_cast(triangles), codes, data, bound, triangle_count); 889| 816| if (!data) ------------------ | Branch (889:6): [True: 60, False: 756] ------------------ 890| 60| return -2; 891| | 892| 756| return (data == bound) ? 0 : -3; ------------------ | Branch (892:9): [True: 15, False: 741] ------------------ 893| 816|} _Z21meshopt_decodeMeshletIjjEiPT_mPT0_mPKhm: 1504| 2.15k|{ 1505| 2.15k| char types_valid[(sizeof(V) == 2 || sizeof(V) == 4) && (sizeof(T) == 1 || sizeof(T) == 4) ? 1 : -1]; 1506| 2.15k| (void)types_valid; 1507| | 1508| 2.15k| return meshopt_decodeMeshlet(vertices, vertex_count, sizeof(V), triangles, triangle_count, sizeof(T) == 1 ? 3 : 4, buffer, buffer_size); ------------------ | Branch (1508:93): [Folded, False: 2.15k] ------------------ 1509| 2.15k|} _Z21meshopt_decodeMeshletIjhEiPT_mPT0_mPKhm: 1504| 4.31k|{ 1505| 4.31k| char types_valid[(sizeof(V) == 2 || sizeof(V) == 4) && (sizeof(T) == 1 || sizeof(T) == 4) ? 1 : -1]; 1506| 4.31k| (void)types_valid; 1507| | 1508| 4.31k| return meshopt_decodeMeshlet(vertices, vertex_count, sizeof(V), triangles, triangle_count, sizeof(T) == 1 ? 3 : 4, buffer, buffer_size); ------------------ | Branch (1508:93): [True: 4.31k, Folded] ------------------ 1509| 4.31k|} _Z21meshopt_decodeMeshletIthEiPT_mPT0_mPKhm: 1504| 2.15k|{ 1505| 2.15k| char types_valid[(sizeof(V) == 2 || sizeof(V) == 4) && (sizeof(T) == 1 || sizeof(T) == 4) ? 1 : -1]; 1506| 2.15k| (void)types_valid; 1507| | 1508| 2.15k| return meshopt_decodeMeshlet(vertices, vertex_count, sizeof(V), triangles, triangle_count, sizeof(T) == 1 ? 3 : 4, buffer, buffer_size); ------------------ | Branch (1508:93): [True: 2.15k, Folded] ------------------ 1509| 2.15k|} meshopt_encodeVertexBufferLevel: 1616| 17.2k|{ 1617| 17.2k| using namespace meshopt; 1618| | 1619| 17.2k| assert(vertex_size > 0 && vertex_size <= 256); ------------------ | Branch (1619:2): [True: 17.2k, False: 0] | Branch (1619:2): [True: 17.2k, False: 0] | Branch (1619:2): [True: 17.2k, False: 0] ------------------ 1620| 17.2k| assert(vertex_size % 4 == 0); ------------------ | Branch (1620:2): [True: 17.2k, False: 0] ------------------ 1621| 17.2k| assert(level >= 0 && level <= 9); // only a subset of this range is used right now ------------------ | Branch (1621:2): [True: 17.2k, False: 0] | Branch (1621:2): [True: 17.2k, False: 0] | Branch (1621:2): [True: 17.2k, False: 0] ------------------ 1622| 17.2k| assert(version < 0 || unsigned(version) <= kDecodeVertexVersion); ------------------ | Branch (1622:2): [True: 17.2k, False: 0] | Branch (1622:2): [True: 0, False: 0] | Branch (1622:2): [True: 17.2k, False: 0] ------------------ 1623| | 1624| 17.2k| version = version < 0 ? gEncodeVertexVersion : version; ------------------ | Branch (1624:12): [True: 17.2k, False: 0] ------------------ 1625| | 1626| |#if TRACE 1627| | memset(vertexstats, 0, sizeof(vertexstats)); 1628| |#endif 1629| | 1630| 17.2k| const unsigned char* vertex_data = static_cast(vertices); 1631| | 1632| 17.2k| unsigned char* data = buffer; 1633| 17.2k| unsigned char* data_end = buffer + buffer_size; 1634| | 1635| 17.2k| if (size_t(data_end - data) < 1) ------------------ | Branch (1635:6): [True: 0, False: 17.2k] ------------------ 1636| 0| return 0; 1637| | 1638| 17.2k| *data++ = (unsigned char)(kVertexHeader | version); 1639| | 1640| 17.2k| unsigned char first_vertex[256] = {}; 1641| 17.2k| if (vertex_count > 0) ------------------ | Branch (1641:6): [True: 13.3k, False: 3.90k] ------------------ 1642| 13.3k| memcpy(first_vertex, vertex_data, vertex_size); 1643| | 1644| 17.2k| unsigned char last_vertex[256] = {}; 1645| 17.2k| memcpy(last_vertex, first_vertex, vertex_size); 1646| | 1647| 17.2k| size_t vertex_block_size = getVertexBlockSize(vertex_size); 1648| | 1649| 17.2k| unsigned char channels[64] = {}; 1650| 17.2k| if (version != 0 && level > 1 && vertex_count > 1) ------------------ | Branch (1650:6): [True: 13.4k, False: 3.81k] | Branch (1650:22): [True: 6.88k, False: 6.56k] | Branch (1650:35): [True: 4.58k, False: 2.29k] ------------------ 1651| 23.8k| for (size_t k = 0; k < vertex_size; k += 4) ------------------ | Branch (1651:22): [True: 19.3k, False: 4.58k] ------------------ 1652| 19.3k| { 1653| 19.3k| int rot = level >= 3 ? estimateRotate(vertex_data, vertex_count, vertex_size, k, /* group_size= */ 16) : 0; ------------------ | Branch (1653:14): [True: 16.0k, False: 3.23k] ------------------ 1654| 19.3k| int channel = estimateChannel(vertex_data, vertex_count, vertex_size, k, vertex_block_size, /* block_skip= */ 3, /* max_channel= */ level >= 3 ? 3 : 2, rot); ------------------ | Branch (1654:136): [True: 16.0k, False: 3.23k] ------------------ 1655| | 1656| 19.3k| assert(unsigned(channel) < 2 || ((channel & 3) == 2 && unsigned(channel >> 4) < 8)); ------------------ | Branch (1656:4): [True: 2.28k, False: 0] | Branch (1656:4): [True: 2.28k, False: 0] | Branch (1656:4): [True: 17.0k, False: 2.28k] | Branch (1656:4): [True: 19.3k, False: 0] ------------------ 1657| 19.3k| channels[k / 4] = (unsigned char)channel; 1658| 19.3k| } 1659| | 1660| 17.2k| size_t vertex_offset = 0; 1661| | 1662| 368k| while (vertex_offset < vertex_count) ------------------ | Branch (1662:9): [True: 351k, False: 17.2k] ------------------ 1663| 351k| { 1664| 351k| size_t block_size = (vertex_offset + vertex_block_size < vertex_count) ? vertex_block_size : vertex_count - vertex_offset; ------------------ | Branch (1664:23): [True: 338k, False: 13.3k] ------------------ 1665| | 1666| 351k| data = encodeVertexBlock(data, data_end, vertex_data + vertex_offset * vertex_size, block_size, vertex_size, last_vertex, channels, version, level); 1667| 351k| if (!data) ------------------ | Branch (1667:7): [True: 0, False: 351k] ------------------ 1668| 0| return 0; 1669| | 1670| 351k| vertex_offset += block_size; 1671| 351k| } 1672| | 1673| 17.2k| size_t tail_size = vertex_size + (version == 0 ? 0 : vertex_size / 4); ------------------ | Branch (1673:36): [True: 3.81k, False: 13.4k] ------------------ 1674| 17.2k| size_t tail_size_min = version == 0 ? kTailMinSizeV0 : kTailMinSizeV1; ------------------ | Branch (1674:25): [True: 3.81k, False: 13.4k] ------------------ 1675| 17.2k| size_t tail_size_pad = tail_size < tail_size_min ? tail_size_min : tail_size; ------------------ | Branch (1675:25): [True: 9.58k, False: 7.67k] ------------------ 1676| | 1677| 17.2k| if (size_t(data_end - data) < tail_size_pad) ------------------ | Branch (1677:6): [True: 0, False: 17.2k] ------------------ 1678| 0| return 0; 1679| | 1680| 17.2k| if (tail_size < tail_size_pad) ------------------ | Branch (1680:6): [True: 9.58k, False: 7.67k] ------------------ 1681| 9.58k| { 1682| 9.58k| memset(data, 0, tail_size_pad - tail_size); 1683| 9.58k| data += tail_size_pad - tail_size; 1684| 9.58k| } 1685| | 1686| 17.2k| memcpy(data, first_vertex, vertex_size); 1687| 17.2k| data += vertex_size; 1688| | 1689| 17.2k| if (version != 0) ------------------ | Branch (1689:6): [True: 13.4k, False: 3.81k] ------------------ 1690| 13.4k| { 1691| 13.4k| memcpy(data, channels, vertex_size / 4); 1692| 13.4k| data += vertex_size / 4; 1693| 13.4k| } 1694| | 1695| 17.2k| assert(data >= buffer + tail_size); ------------------ | Branch (1695:2): [True: 17.2k, False: 0] ------------------ 1696| 17.2k| assert(data <= buffer + buffer_size); ------------------ | Branch (1696:2): [True: 17.2k, False: 0] ------------------ 1697| | 1698| |#if TRACE 1699| | size_t total_size = data - buffer; 1700| | 1701| | for (size_t k = 0; k < vertex_size; ++k) 1702| | { 1703| | const Stats& vsk = vertexstats[k]; 1704| | 1705| | printf("%2d: %7d bytes [%4.1f%%] %.1f bpv", int(k), int(vsk.size), double(vsk.size) / double(total_size) * 100, double(vsk.size) / double(vertex_count) * 8); 1706| | 1707| | size_t total_k = vsk.header + vsk.bitg[1] + vsk.bitg[2] + vsk.bitg[4] + vsk.bitg[8]; 1708| | double total_kr = total_k ? 1.0 / double(total_k) : 0; 1709| | 1710| | if (version != 0) 1711| | { 1712| | int channel = channels[k / 4]; 1713| | 1714| | if ((channel & 3) == 2 && k % 4 == 0) 1715| | printf(" | ^%d", channel >> 4); 1716| | else 1717| | printf(" | %2s", channel == 0 ? "1" : (channel == 1 && k % 2 == 0 ? "2" : ".")); 1718| | } 1719| | 1720| | printf(" | hdr [%5.1f%%] bitg [1 %4.1f%% 2 %4.1f%% 4 %4.1f%% 8 %4.1f%%]", 1721| | double(vsk.header) * total_kr * 100, 1722| | double(vsk.bitg[1]) * total_kr * 100, double(vsk.bitg[2]) * total_kr * 100, 1723| | double(vsk.bitg[4]) * total_kr * 100, double(vsk.bitg[8]) * total_kr * 100); 1724| | 1725| | size_t total_ctrl = vsk.ctrl[0] + vsk.ctrl[1] + vsk.ctrl[2] + vsk.ctrl[3]; 1726| | 1727| | if (total_ctrl) 1728| | { 1729| | printf(" | ctrl %3.0f%% %3.0f%% %3.0f%% %3.0f%%", 1730| | double(vsk.ctrl[0]) / double(total_ctrl) * 100, double(vsk.ctrl[1]) / double(total_ctrl) * 100, 1731| | double(vsk.ctrl[2]) / double(total_ctrl) * 100, double(vsk.ctrl[3]) / double(total_ctrl) * 100); 1732| | } 1733| | 1734| | if (level >= 3) 1735| | printf(" | bitc [%3.0f%% %3.0f%% %3.0f%% %3.0f%% %3.0f%% %3.0f%% %3.0f%% %3.0f%%]", 1736| | double(vsk.bitc[0]) / double(vertex_count) * 100, double(vsk.bitc[1]) / double(vertex_count) * 100, 1737| | double(vsk.bitc[2]) / double(vertex_count) * 100, double(vsk.bitc[3]) / double(vertex_count) * 100, 1738| | double(vsk.bitc[4]) / double(vertex_count) * 100, double(vsk.bitc[5]) / double(vertex_count) * 100, 1739| | double(vsk.bitc[6]) / double(vertex_count) * 100, double(vsk.bitc[7]) / double(vertex_count) * 100); 1740| | 1741| | printf("\n"); 1742| | } 1743| |#endif 1744| | 1745| 17.2k| return data - buffer; 1746| 17.2k|} meshopt_encodeVertexBufferBound: 1754| 8.62k|{ 1755| 8.62k| using namespace meshopt; 1756| | 1757| 8.62k| assert(vertex_size > 0 && vertex_size <= 256); ------------------ | Branch (1757:2): [True: 8.62k, False: 0] | Branch (1757:2): [True: 8.62k, False: 0] | Branch (1757:2): [True: 8.62k, False: 0] ------------------ 1758| 8.62k| assert(vertex_size % 4 == 0); ------------------ | Branch (1758:2): [True: 8.62k, False: 0] ------------------ 1759| | 1760| 8.62k| size_t vertex_block_size = getVertexBlockSize(vertex_size); 1761| 8.62k| size_t vertex_block_count = (vertex_count + vertex_block_size - 1) / vertex_block_size; 1762| | 1763| 8.62k| size_t vertex_block_control_size = vertex_size / 4; 1764| 8.62k| size_t vertex_byte_header_size = (vertex_block_size / kByteGroupSize + 3) / 4; 1765| 8.62k| size_t vertex_byte_data_size = vertex_block_size; 1766| | 1767| 8.62k| size_t tail_size = vertex_size + (vertex_size / 4); 1768| 8.62k| size_t tail_size_min = kTailMinSizeV0 > kTailMinSizeV1 ? kTailMinSizeV0 : kTailMinSizeV1; ------------------ | Branch (1768:25): [True: 8.62k, Folded] ------------------ 1769| 8.62k| size_t tail_size_pad = tail_size < tail_size_min ? tail_size_min : tail_size; ------------------ | Branch (1769:25): [True: 6.47k, False: 2.15k] ------------------ 1770| 8.62k| assert(tail_size_pad >= kByteGroupDecodeLimit); ------------------ | Branch (1770:2): [True: 8.62k, False: 0] ------------------ 1771| | 1772| 8.62k| return 1 + vertex_block_count * (vertex_block_control_size + vertex_size * (vertex_byte_header_size + vertex_byte_data_size)) + tail_size_pad; 1773| 8.62k|} meshopt_encodeVertexVersion: 1776| 2.15k|{ 1777| 2.15k| assert(unsigned(version) <= unsigned(meshopt::kDecodeVertexVersion)); ------------------ | Branch (1777:2): [True: 2.15k, False: 0] ------------------ 1778| | 1779| 2.15k| meshopt::gEncodeVertexVersion = version; 1780| 2.15k|} meshopt_decodeVertexBuffer: 1800| 17.2k|{ 1801| 17.2k| using namespace meshopt; 1802| | 1803| 17.2k| assert(vertex_size > 0 && vertex_size <= 256); ------------------ | Branch (1803:2): [True: 17.2k, False: 0] | Branch (1803:2): [True: 17.2k, False: 0] | Branch (1803:2): [True: 17.2k, False: 0] ------------------ 1804| 17.2k| assert(vertex_size % 4 == 0); ------------------ | Branch (1804:2): [True: 17.2k, False: 0] ------------------ 1805| | 1806| 17.2k| const unsigned char* (*decode)(const unsigned char*, const unsigned char*, unsigned char*, size_t, size_t, unsigned char[256], const unsigned char*, int) = NULL; 1807| | 1808| 17.2k|#if defined(SIMD_SSE) && defined(SIMD_FALLBACK) 1809| 17.2k| const unsigned int cpumask = (1 << 9) | (1 << 23); // SSSE3+POPCNT 1810| 17.2k| decode = (cpuid & cpumask) == cpumask ? decodeVertexBlockSimd : decodeVertexBlock; ------------------ | Branch (1810:11): [True: 17.2k, False: 0] ------------------ 1811| |#elif defined(SIMD_SSE) || defined(SIMD_AVX) || defined(SIMD_NEON) || defined(SIMD_WASM) 1812| | decode = decodeVertexBlockSimd; 1813| |#else 1814| | decode = decodeVertexBlock; 1815| |#endif 1816| | 1817| 17.2k|#if defined(SIMD_SSE) || defined(SIMD_NEON) || defined(SIMD_WASM) 1818| 17.2k| assert(gDecodeBytesGroupInitialized); ------------------ | Branch (1818:2): [True: 17.2k, False: 0] ------------------ 1819| 17.2k| (void)gDecodeBytesGroupInitialized; 1820| 17.2k|#endif 1821| | 1822| 17.2k| unsigned char* vertex_data = static_cast(destination); 1823| | 1824| 17.2k| const unsigned char* data = buffer; 1825| 17.2k| const unsigned char* data_end = buffer + buffer_size; 1826| | 1827| 17.2k| if (size_t(data_end - data) < 1) ------------------ | Branch (1827:6): [True: 0, False: 17.2k] ------------------ 1828| 0| return -2; 1829| | 1830| 17.2k| unsigned char data_header = *data++; 1831| | 1832| 17.2k| if ((data_header & 0xf0) != kVertexHeader) ------------------ | Branch (1832:6): [True: 7.37k, False: 9.88k] ------------------ 1833| 7.37k| return -1; 1834| | 1835| 9.88k| int version = data_header & 0x0f; 1836| 9.88k| if (version > kDecodeVertexVersion) ------------------ | Branch (1836:6): [True: 116, False: 9.76k] ------------------ 1837| 116| return -1; 1838| | 1839| 9.76k| size_t tail_size = vertex_size + (version == 0 ? 0 : vertex_size / 4); ------------------ | Branch (1839:36): [True: 2.21k, False: 7.55k] ------------------ 1840| 9.76k| size_t tail_size_min = version == 0 ? kTailMinSizeV0 : kTailMinSizeV1; ------------------ | Branch (1840:25): [True: 2.21k, False: 7.55k] ------------------ 1841| 9.76k| size_t tail_size_pad = tail_size < tail_size_min ? tail_size_min : tail_size; ------------------ | Branch (1841:25): [True: 5.43k, False: 4.32k] ------------------ 1842| | 1843| 9.76k| if (size_t(data_end - data) < tail_size_pad) ------------------ | Branch (1843:6): [True: 157, False: 9.60k] ------------------ 1844| 157| return -2; 1845| | 1846| 9.60k| const unsigned char* tail = data_end - tail_size; 1847| | 1848| 9.60k| unsigned char last_vertex[256]; 1849| 9.60k| memcpy(last_vertex, tail, vertex_size); 1850| | 1851| 9.60k| const unsigned char* channels = version == 0 ? NULL : tail + vertex_size; ------------------ | Branch (1851:34): [True: 2.17k, False: 7.43k] ------------------ 1852| | 1853| 9.60k| size_t vertex_block_size = getVertexBlockSize(vertex_size); 1854| | 1855| 9.60k| size_t vertex_offset = 0; 1856| | 1857| 185k| while (vertex_offset < vertex_count) ------------------ | Branch (1857:9): [True: 176k, False: 9.11k] ------------------ 1858| 176k| { 1859| 176k| size_t block_size = (vertex_offset + vertex_block_size < vertex_count) ? vertex_block_size : vertex_count - vertex_offset; ------------------ | Branch (1859:23): [True: 169k, False: 7.65k] ------------------ 1860| | 1861| 176k| data = decode(data, data_end, vertex_data + vertex_offset * vertex_size, block_size, vertex_size, last_vertex, channels, version); 1862| 176k| if (!data) ------------------ | Branch (1862:7): [True: 494, False: 176k] ------------------ 1863| 494| return -2; 1864| | 1865| 176k| vertex_offset += block_size; 1866| 176k| } 1867| | 1868| 9.11k| if (size_t(data_end - data) != tail_size_pad) ------------------ | Branch (1868:6): [True: 484, False: 8.62k] ------------------ 1869| 484| return -3; 1870| | 1871| 8.62k| return 0; 1872| 9.11k|} vertexcodec.cpp:_ZN7meshoptL27decodeBytesGroupBuildTablesEv: 792| 2|{ 793| 514| for (int mask = 0; mask < 256; ++mask) ------------------ | Branch (793:21): [True: 512, False: 2] ------------------ 794| 512| { 795| 512| unsigned char shuffle[8]; 796| 512| unsigned char count = 0; 797| | 798| 4.60k| for (int i = 0; i < 8; ++i) ------------------ | Branch (798:19): [True: 4.09k, False: 512] ------------------ 799| 4.09k| { 800| 4.09k| int maski = (mask >> i) & 1; 801| 4.09k| shuffle[i] = maski ? count : 0x80; ------------------ | Branch (801:17): [True: 2.04k, False: 2.04k] ------------------ 802| 4.09k| count += (unsigned char)(maski); 803| 4.09k| } 804| | 805| 512| memcpy(kDecodeBytesGroupShuffle[mask], shuffle, 8); 806| 512| kDecodeBytesGroupCount[mask] = count; 807| 512| } 808| | 809| 2| return true; 810| 2|} vertexcodec.cpp:_ZN7meshoptL14getCpuFeaturesEv: 1600| 2|{ 1601| 2| int cpuinfo[4] = {}; 1602| |#ifdef _MSC_VER 1603| | __cpuid(cpuinfo, 1); 1604| |#else 1605| | __cpuid(1, cpuinfo[0], cpuinfo[1], cpuinfo[2], cpuinfo[3]); 1606| 2|#endif 1607| 2| return cpuinfo[2]; 1608| 2|} vertexcodec.cpp:_ZN7meshoptL18getVertexBlockSizeEm: 141| 35.4k|{ 142| | // make sure the entire block fits into the scratch buffer and is aligned to byte group size 143| | // note: the block size is implicitly part of the format, so we can't change it without breaking compatibility 144| 35.4k| size_t result = (kVertexBlockSizeBytes / vertex_size) & ~(kByteGroupSize - 1); 145| | 146| 35.4k| return (result < kVertexBlockMaxSize) ? result : kVertexBlockMaxSize; ------------------ | Branch (146:9): [True: 0, False: 35.4k] ------------------ 147| 35.4k|} vertexcodec.cpp:_ZN7meshoptL14estimateRotateEPKhmmmm: 370| 16.0k|{ 371| 16.0k| size_t sizes[8] = {}; 372| | 373| 16.0k| const unsigned char* vertex = vertex_data + k; 374| 16.0k| unsigned int last = vertex[0] | (vertex[1] << 8) | (vertex[2] << 16) | (vertex[3] << 24); 375| | 376| 4.48M| for (size_t i = 0; i < vertex_count; i += group_size) ------------------ | Branch (376:21): [True: 4.47M, False: 16.0k] ------------------ 377| 4.47M| { 378| 4.47M| unsigned int bitg = 0; 379| | 380| | // calculate bit consistency mask for the group 381| 75.9M| for (size_t j = 0; j < group_size && i + j < vertex_count; ++j) ------------------ | Branch (381:22): [True: 71.4M, False: 4.45M] | Branch (381:40): [True: 71.4M, False: 13.9k] ------------------ 382| 71.4M| { 383| 71.4M| unsigned int v = vertex[0] | (vertex[1] << 8) | (vertex[2] << 16) | (vertex[3] << 24); 384| 71.4M| unsigned int d = v ^ last; 385| | 386| 71.4M| bitg |= d; 387| 71.4M| last = v; 388| 71.4M| vertex += vertex_size; 389| 71.4M| } 390| | 391| |#if TRACE 392| | for (int j = 0; j < 32; ++j) 393| | vertexstats[k + (j / 8)].bitc[j % 8] += (i + group_size < vertex_count ? group_size : vertex_count - i) * (1 - ((bitg >> j) & 1)); 394| |#endif 395| | 396| 40.2M| for (int j = 0; j < 8; ++j) ------------------ | Branch (396:19): [True: 35.7M, False: 4.47M] ------------------ 397| 35.7M| { 398| 35.7M| unsigned int bitr = rotate(bitg, j); 399| | 400| 35.7M| sizes[j] += estimateBits((unsigned char)(bitr >> 0)) + estimateBits((unsigned char)(bitr >> 8)); 401| 35.7M| sizes[j] += estimateBits((unsigned char)(bitr >> 16)) + estimateBits((unsigned char)(bitr >> 24)); 402| 35.7M| } 403| 4.47M| } 404| | 405| 16.0k| int best_rot = 0; 406| 128k| for (int rot = 1; rot < 8; ++rot) ------------------ | Branch (406:20): [True: 112k, False: 16.0k] ------------------ 407| 112k| best_rot = (sizes[rot] < sizes[best_rot]) ? rot : best_rot; ------------------ | Branch (407:14): [True: 6.23k, False: 106k] ------------------ 408| | 409| 16.0k| return best_rot; 410| 16.0k|} _ZN7meshopt6rotateEji: 150| 173M|{ 151| 173M| return (v << r) | (v >> ((32 - r) & 31)); 152| 173M|} vertexcodec.cpp:_ZN7meshoptL12estimateBitsEh: 365| 143M|{ 366| 143M| return v <= 15 ? (v <= 3 ? (v == 0 ? 0 : 2) : 4) : 8; ------------------ | Branch (366:9): [True: 55.3M, False: 87.8M] | Branch (366:20): [True: 52.6M, False: 2.63M] | Branch (366:30): [True: 50.3M, False: 2.37M] ------------------ 367| 143M|} vertexcodec.cpp:_ZN7meshoptL15estimateChannelEPKhmmmmmii: 413| 19.3k|{ 414| 19.3k| unsigned char block[kVertexBlockMaxSize]; 415| 19.3k| assert(vertex_block_size <= kVertexBlockMaxSize); ------------------ | Branch (415:2): [True: 19.3k, False: 0] ------------------ 416| | 417| 19.3k| unsigned char last_vertex[256] = {}; 418| | 419| 19.3k| size_t sizes[3] = {}; 420| 19.3k| assert(max_channel <= 3); ------------------ | Branch (420:2): [True: 19.3k, False: 0] ------------------ 421| | 422| 153k| for (size_t i = 0; i < vertex_count; i += vertex_block_size * block_skip) ------------------ | Branch (422:21): [True: 134k, False: 19.3k] ------------------ 423| 134k| { 424| 134k| size_t block_size = i + vertex_block_size < vertex_count ? vertex_block_size : vertex_count - i; ------------------ | Branch (424:23): [True: 119k, False: 14.9k] ------------------ 425| 134k| size_t block_size_aligned = (block_size + kByteGroupSize - 1) & ~(kByteGroupSize - 1); 426| | 427| 134k| memcpy(last_vertex, vertex_data + (i == 0 ? 0 : i - 1) * vertex_size, vertex_size); ------------------ | Branch (427:38): [True: 19.3k, False: 114k] ------------------ 428| | 429| | // we sometimes encode elements we didn't fill when rounding to kByteGroupSize 430| 134k| if (block_size < block_size_aligned) ------------------ | Branch (430:7): [True: 13.2k, False: 120k] ------------------ 431| 13.2k| memset(block + block_size, 0, block_size_aligned - block_size); 432| | 433| 508k| for (int channel = 0; channel < max_channel; ++channel) ------------------ | Branch (433:25): [True: 374k, False: 134k] ------------------ 434| 1.87M| for (size_t j = 0; j < 4; ++j) ------------------ | Branch (434:23): [True: 1.49M, False: 374k] ------------------ 435| 1.49M| { 436| 1.49M| encodeDeltas(block, vertex_data + i * vertex_size, block_size, vertex_size, last_vertex, k + j, channel | (xor_rot << 4)); 437| | 438| 23.3M| for (size_t ig = 0; ig < block_size; ig += kByteGroupSize) ------------------ | Branch (438:25): [True: 21.8M, False: 1.49M] ------------------ 439| 21.8M| { 440| | // to maximize encoding performance we only evaluate 1/2/4/8 bit groups 441| 21.8M| size_t size1 = encodeBytesGroupMeasure(block + ig, 1); 442| 21.8M| size_t size2 = encodeBytesGroupMeasure(block + ig, 2); 443| 21.8M| size_t size4 = encodeBytesGroupMeasure(block + ig, 4); 444| 21.8M| size_t size8 = encodeBytesGroupMeasure(block + ig, 8); 445| | 446| 21.8M| size_t best_size = size1 < size2 ? size1 : size2; ------------------ | Branch (446:25): [True: 20.4M, False: 1.44M] ------------------ 447| 21.8M| best_size = best_size < size4 ? best_size : size4; ------------------ | Branch (447:18): [True: 21.8M, False: 62.6k] ------------------ 448| 21.8M| best_size = best_size < size8 ? best_size : size8; ------------------ | Branch (448:18): [True: 11.7M, False: 10.0M] ------------------ 449| | 450| 21.8M| sizes[channel] += best_size; 451| 21.8M| } 452| 1.49M| } 453| 134k| } 454| | 455| 19.3k| int best_channel = 0; 456| 54.6k| for (int channel = 1; channel < max_channel; ++channel) ------------------ | Branch (456:24): [True: 35.3k, False: 19.3k] ------------------ 457| 35.3k| best_channel = (sizes[channel] < sizes[best_channel]) ? channel : best_channel; ------------------ | Branch (457:18): [True: 5.43k, False: 29.9k] ------------------ 458| | 459| 19.3k| return best_channel == 2 ? best_channel | (xor_rot << 4) : best_channel; ------------------ | Branch (459:9): [True: 2.28k, False: 17.0k] ------------------ 460| 19.3k|} vertexcodec.cpp:_ZN7meshoptL12encodeDeltasEPhPKhmmS2_mi: 350| 5.21M|{ 351| 5.21M| switch (channel & 3) 352| 5.21M| { 353| 3.98M| case 0: ------------------ | Branch (353:2): [True: 3.98M, False: 1.23M] ------------------ 354| 3.98M| return encodeDeltas1(buffer, vertex_data, vertex_count, vertex_size, last_vertex, k, 0); 355| 658k| case 1: ------------------ | Branch (355:2): [True: 658k, False: 4.56M] ------------------ 356| 658k| return encodeDeltas1(buffer, vertex_data, vertex_count, vertex_size, last_vertex, k, 0); 357| 581k| case 2: ------------------ | Branch (357:2): [True: 581k, False: 4.63M] ------------------ 358| 581k| return encodeDeltas1(buffer, vertex_data, vertex_count, vertex_size, last_vertex, k, channel >> 4); 359| 0| default: ------------------ | Branch (359:2): [True: 0, False: 5.21M] ------------------ 360| | assert(!"Unsupported channel encoding"); // unreachable ------------------ | Branch (360:3): [Folded, False: 0] ------------------ 361| 5.21M| } 362| 5.21M|} vertexcodec.cpp:_ZN7meshoptL13encodeDeltas1IhLb0EEEvPhPKhmmS3_mi: 325| 3.98M|{ 326| 3.98M| size_t k0 = k & ~(sizeof(T) - 1); 327| 3.98M| int ks = (k & (sizeof(T) - 1)) * 8; 328| | 329| 3.98M| T p = last_vertex[k0]; 330| 3.98M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (330:21): [True: 0, False: 3.98M] ------------------ 331| 0| p |= T(last_vertex[k0 + j]) << (j * 8); 332| | 333| 3.98M| const unsigned char* vertex = vertex_data + k0; 334| | 335| 966M| for (size_t i = 0; i < vertex_count; ++i) ------------------ | Branch (335:21): [True: 962M, False: 3.98M] ------------------ 336| 962M| { 337| 962M| T v = vertex[0]; 338| 962M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (338:22): [True: 0, False: 962M] ------------------ 339| 0| v |= vertex[j] << (j * 8); 340| | 341| 962M| T d = Xor ? T(rotate(v ^ p, rot)) : zigzag(T(v - p)); ------------------ | Branch (341:9): [Folded, False: 962M] ------------------ 342| | 343| 962M| buffer[i] = (unsigned char)(d >> ks); 344| 962M| p = v; 345| 962M| vertex += vertex_size; 346| 962M| } 347| 3.98M|} _ZN7meshopt6zigzagIhEET_S1_: 156| 962M|{ 157| 962M| return (0 - (v >> (sizeof(T) * 8 - 1))) ^ (v << 1); 158| 962M|} vertexcodec.cpp:_ZN7meshoptL13encodeDeltas1ItLb0EEEvPhPKhmmS3_mi: 325| 658k|{ 326| 658k| size_t k0 = k & ~(sizeof(T) - 1); 327| 658k| int ks = (k & (sizeof(T) - 1)) * 8; 328| | 329| 658k| T p = last_vertex[k0]; 330| 1.31M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (330:21): [True: 658k, False: 658k] ------------------ 331| 658k| p |= T(last_vertex[k0 + j]) << (j * 8); 332| | 333| 658k| const unsigned char* vertex = vertex_data + k0; 334| | 335| 154M| for (size_t i = 0; i < vertex_count; ++i) ------------------ | Branch (335:21): [True: 153M, False: 658k] ------------------ 336| 153M| { 337| 153M| T v = vertex[0]; 338| 307M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (338:22): [True: 153M, False: 153M] ------------------ 339| 153M| v |= vertex[j] << (j * 8); 340| | 341| 153M| T d = Xor ? T(rotate(v ^ p, rot)) : zigzag(T(v - p)); ------------------ | Branch (341:9): [Folded, False: 153M] ------------------ 342| | 343| 153M| buffer[i] = (unsigned char)(d >> ks); 344| 153M| p = v; 345| 153M| vertex += vertex_size; 346| 153M| } 347| 658k|} _ZN7meshopt6zigzagItEET_S1_: 156| 153M|{ 157| 153M| return (0 - (v >> (sizeof(T) * 8 - 1))) ^ (v << 1); 158| 153M|} vertexcodec.cpp:_ZN7meshoptL13encodeDeltas1IjLb1EEEvPhPKhmmS3_mi: 325| 581k|{ 326| 581k| size_t k0 = k & ~(sizeof(T) - 1); 327| 581k| int ks = (k & (sizeof(T) - 1)) * 8; 328| | 329| 581k| T p = last_vertex[k0]; 330| 2.32M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (330:21): [True: 1.74M, False: 581k] ------------------ 331| 1.74M| p |= T(last_vertex[k0 + j]) << (j * 8); 332| | 333| 581k| const unsigned char* vertex = vertex_data + k0; 334| | 335| 137M| for (size_t i = 0; i < vertex_count; ++i) ------------------ | Branch (335:21): [True: 137M, False: 581k] ------------------ 336| 137M| { 337| 137M| T v = vertex[0]; 338| 549M| for (size_t j = 1; j < sizeof(T); ++j) ------------------ | Branch (338:22): [True: 411M, False: 137M] ------------------ 339| 411M| v |= vertex[j] << (j * 8); 340| | 341| 137M| T d = Xor ? T(rotate(v ^ p, rot)) : zigzag(T(v - p)); ------------------ | Branch (341:9): [True: 137M, Folded] ------------------ 342| | 343| 137M| buffer[i] = (unsigned char)(d >> ks); 344| 137M| p = v; 345| 137M| vertex += vertex_size; 346| 137M| } 347| 581k|} vertexcodec.cpp:_ZN7meshoptL23encodeBytesGroupMeasureEPKhi: 191| 286M|{ 192| 286M| assert(bits >= 0 && bits <= 8); ------------------ | Branch (192:2): [True: 286M, False: 0] | Branch (192:2): [True: 286M, False: 0] | Branch (192:2): [True: 286M, False: 0] ------------------ 193| | 194| 286M| if (bits == 0) ------------------ | Branch (194:6): [True: 27.3M, False: 259M] ------------------ 195| 27.3M| return encodeBytesGroupZero(buffer) ? 0 : size_t(-1); ------------------ | Branch (195:10): [True: 5.71M, False: 21.6M] ------------------ 196| | 197| 259M| if (bits == 8) ------------------ | Branch (197:6): [True: 63.2M, False: 196M] ------------------ 198| 63.2M| return kByteGroupSize; 199| | 200| 196M| size_t result = kByteGroupSize * bits / 8; 201| | 202| 196M| unsigned char sentinel = (1 << bits) - 1; 203| | 204| 3.33G| for (size_t i = 0; i < kByteGroupSize; ++i) ------------------ | Branch (204:21): [True: 3.13G, False: 196M] ------------------ 205| 3.13G| result += buffer[i] >= sentinel; 206| | 207| 196M| return result; 208| 259M|} vertexcodec.cpp:_ZN7meshoptL20encodeBytesGroupZeroEPKh: 181| 49.7M|{ 182| 49.7M| assert(kByteGroupSize == sizeof(unsigned long long) * 2); ------------------ | Branch (182:2): [True: 49.7M, Folded] ------------------ 183| | 184| 49.7M| unsigned long long v[2]; 185| 49.7M| memcpy(v, buffer, sizeof(v)); 186| | 187| 49.7M| return (v[0] | v[1]) == 0; 188| 49.7M|} vertexcodec.cpp:_ZN7meshoptL17encodeVertexBlockEPhS0_PKhmmS0_S2_ii: 510| 351k|{ 511| 351k| assert(vertex_count > 0 && vertex_count <= kVertexBlockMaxSize); ------------------ | Branch (511:2): [True: 351k, False: 0] | Branch (511:2): [True: 351k, False: 0] | Branch (511:2): [True: 351k, False: 0] ------------------ 512| 351k| assert(vertex_size % 4 == 0); ------------------ | Branch (512:2): [True: 351k, False: 0] ------------------ 513| | 514| 351k| unsigned char buffer[kVertexBlockMaxSize]; 515| 351k| assert(sizeof(buffer) % kByteGroupSize == 0); ------------------ | Branch (515:2): [True: 351k, Folded] ------------------ 516| | 517| 351k| size_t vertex_count_aligned = (vertex_count + kByteGroupSize - 1) & ~(kByteGroupSize - 1); 518| | 519| | // we sometimes encode elements we didn't fill when rounding to kByteGroupSize 520| 351k| memset(buffer, 0, sizeof(buffer)); 521| | 522| 351k| size_t control_size = version == 0 ? 0 : vertex_size / 4; ------------------ | Branch (522:24): [True: 20.8k, False: 330k] ------------------ 523| 351k| if (size_t(data_end - data) < control_size) ------------------ | Branch (523:6): [True: 0, False: 351k] ------------------ 524| 0| return NULL; 525| | 526| 351k| unsigned char* control = data; 527| 351k| data += control_size; 528| | 529| 351k| memset(control, 0, control_size); 530| | 531| 4.07M| for (size_t k = 0; k < vertex_size; ++k) ------------------ | Branch (531:21): [True: 3.72M, False: 351k] ------------------ 532| 3.72M| { 533| 3.72M| encodeDeltas(buffer, vertex_data, vertex_count, vertex_size, last_vertex, k, version == 0 ? 0 : channels[k / 4]); ------------------ | Branch (533:80): [True: 238k, False: 3.48M] ------------------ 534| | 535| |#if TRACE 536| | const unsigned char* olddata = data; 537| | bytestats = &vertexstats[k]; 538| |#endif 539| | 540| 3.72M| int ctrl = 0; 541| | 542| 3.72M| if (version != 0) ------------------ | Branch (542:7): [True: 3.48M, False: 238k] ------------------ 543| 3.48M| { 544| 3.48M| ctrl = estimateControl(buffer, vertex_count, vertex_count_aligned, level); 545| | 546| 3.48M| assert(unsigned(ctrl) < 4); ------------------ | Branch (546:4): [True: 3.48M, False: 0] ------------------ 547| 3.48M| control[k / 4] |= ctrl << ((k % 4) * 2); 548| | 549| |#if TRACE 550| | vertexstats[k].ctrl[ctrl]++; 551| |#endif 552| 3.48M| } 553| | 554| 3.72M| if (ctrl == 3) ------------------ | Branch (554:7): [True: 919k, False: 2.80M] ------------------ 555| 919k| { 556| | // literal encoding 557| 919k| if (size_t(data_end - data) < vertex_count) ------------------ | Branch (557:8): [True: 0, False: 919k] ------------------ 558| 0| return NULL; 559| | 560| 919k| memcpy(data, buffer, vertex_count); 561| 919k| data += vertex_count; 562| 919k| } 563| 2.80M| else if (ctrl != 2) // non-zero encoding ------------------ | Branch (563:12): [True: 1.54M, False: 1.25M] ------------------ 564| 1.54M| { 565| 1.54M| data = encodeBytes(data, data_end, buffer, vertex_count_aligned, version == 0 ? kBitsV0 : kBitsV1 + ctrl); ------------------ | Branch (565:69): [True: 238k, False: 1.31M] ------------------ 566| 1.54M| if (!data) ------------------ | Branch (566:8): [True: 0, False: 1.54M] ------------------ 567| 0| return NULL; 568| 1.54M| } 569| | 570| |#if TRACE 571| | bytestats = NULL; 572| | vertexstats[k].size += data - olddata; 573| |#endif 574| 3.72M| } 575| | 576| 351k| memcpy(last_vertex, &vertex_data[vertex_size * (vertex_count - 1)], vertex_size); 577| | 578| 351k| return data; 579| 351k|} vertexcodec.cpp:_ZN7meshoptL15estimateControlEPKhmmi: 472| 3.48M|{ 473| 3.48M| if (estimateControlZero(buffer, vertex_count_aligned)) ------------------ | Branch (473:6): [True: 1.25M, False: 2.23M] ------------------ 474| 1.25M| return 2; // zero encoding 475| | 476| 2.23M| if (level == 0) ------------------ | Branch (476:6): [True: 846k, False: 1.38M] ------------------ 477| 846k| return 1; // 1248 encoding in level 0 for encoding speed 478| | 479| | // round number of groups to 4 to get number of header bytes 480| 1.38M| size_t header_size = (vertex_count_aligned / kByteGroupSize + 3) / 4; 481| | 482| 1.38M| size_t est_bytes0 = header_size, est_bytes1 = header_size; 483| | 484| 22.5M| for (size_t i = 0; i < vertex_count_aligned; i += kByteGroupSize) ------------------ | Branch (484:21): [True: 21.1M, False: 1.38M] ------------------ 485| 21.1M| { 486| | // assumes kBitsV1[] = {0, 1, 2, 4, 8} for performance 487| 21.1M| size_t size0 = encodeBytesGroupMeasure(buffer + i, 0); 488| 21.1M| size_t size1 = encodeBytesGroupMeasure(buffer + i, 1); 489| 21.1M| size_t size2 = encodeBytesGroupMeasure(buffer + i, 2); 490| 21.1M| size_t size4 = encodeBytesGroupMeasure(buffer + i, 4); 491| 21.1M| size_t size8 = encodeBytesGroupMeasure(buffer + i, 8); 492| | 493| | // both control modes have access to 1/2/4 bit encoding 494| 21.1M| size_t size12 = size1 < size2 ? size1 : size2; ------------------ | Branch (494:19): [True: 18.7M, False: 2.44M] ------------------ 495| 21.1M| size_t size124 = size12 < size4 ? size12 : size4; ------------------ | Branch (495:20): [True: 21.1M, False: 59.3k] ------------------ 496| | 497| | // each control mode has access to 0/8 bit encoding respectively 498| 21.1M| est_bytes0 += size124 < size0 ? size124 : size0; ------------------ | Branch (498:17): [True: 18.9M, False: 2.23M] ------------------ 499| 21.1M| est_bytes1 += size124 < size8 ? size124 : size8; ------------------ | Branch (499:17): [True: 6.06M, False: 15.1M] ------------------ 500| 21.1M| } 501| | 502| | // pick shortest control entry but prefer literal encoding 503| 1.38M| if (est_bytes0 < vertex_count || est_bytes1 < vertex_count) ------------------ | Branch (503:6): [True: 419k, False: 965k] | Branch (503:35): [True: 45.0k, False: 919k] ------------------ 504| 464k| return est_bytes0 < est_bytes1 ? 0 : 1; ------------------ | Branch (504:10): [True: 199k, False: 265k] ------------------ 505| 919k| else 506| 919k| return 3; // literal encoding 507| 1.38M|} vertexcodec.cpp:_ZN7meshoptL19estimateControlZeroEPKhm: 463| 3.48M|{ 464| 23.6M| for (size_t i = 0; i < vertex_count_aligned; i += kByteGroupSize) ------------------ | Branch (464:21): [True: 22.4M, False: 1.25M] ------------------ 465| 22.4M| if (!encodeBytesGroupZero(buffer + i)) ------------------ | Branch (465:7): [True: 2.23M, False: 20.2M] ------------------ 466| 2.23M| return false; 467| | 468| 1.25M| return true; 469| 3.48M|} vertexcodec.cpp:_ZN7meshoptL11encodeBytesEPhS0_PKhmPKi: 264| 1.54M|{ 265| 1.54M| assert(buffer_size % kByteGroupSize == 0); ------------------ | Branch (265:2): [True: 1.54M, False: 0] ------------------ 266| | 267| 1.54M| unsigned char* header = data; 268| | 269| | // round number of groups to 4 to get number of header bytes 270| 1.54M| size_t header_size = (buffer_size / kByteGroupSize + 3) / 4; 271| | 272| 1.54M| if (size_t(data_end - data) < header_size) ------------------ | Branch (272:6): [True: 0, False: 1.54M] ------------------ 273| 0| return NULL; 274| | 275| 1.54M| data += header_size; 276| | 277| 1.54M| memset(header, 0, header_size); 278| | 279| 1.54M| int last_bits = -1; 280| | 281| 24.8M| for (size_t i = 0; i < buffer_size; i += kByteGroupSize) ------------------ | Branch (281:21): [True: 23.3M, False: 1.54M] ------------------ 282| 23.3M| { 283| 23.3M| if (size_t(data_end - data) < kByteGroupDecodeLimit) ------------------ | Branch (283:7): [True: 0, False: 23.3M] ------------------ 284| 0| return NULL; 285| | 286| 23.3M| int best_bitk = 3; 287| 23.3M| size_t best_size = encodeBytesGroupMeasure(buffer + i, bits[best_bitk]); 288| | 289| 93.3M| for (int bitk = 0; bitk < 3; ++bitk) ------------------ | Branch (289:22): [True: 70.0M, False: 23.3M] ------------------ 290| 70.0M| { 291| 70.0M| size_t size = encodeBytesGroupMeasure(buffer + i, bits[bitk]); 292| | 293| | // favor consistent bit selection across groups, but never replace literals 294| 70.0M| if (size < best_size || (size == best_size && bits[bitk] == last_bits && bits[best_bitk] != 8)) ------------------ | Branch (294:8): [True: 11.8M, False: 58.1M] | Branch (294:29): [True: 1.10M, False: 57.0M] | Branch (294:50): [True: 271k, False: 837k] | Branch (294:77): [True: 50.5k, False: 221k] ------------------ 295| 11.9M| { 296| 11.9M| best_bitk = bitk; 297| 11.9M| best_size = size; 298| 11.9M| } 299| 70.0M| } 300| | 301| 23.3M| size_t header_offset = i / kByteGroupSize; 302| 23.3M| header[header_offset / 4] |= best_bitk << ((header_offset % 4) * 2); 303| | 304| 23.3M| int best_bits = bits[best_bitk]; 305| 23.3M| unsigned char* next = encodeBytesGroup(data, buffer + i, best_bits); 306| | 307| 23.3M| assert(data + best_size == next); ------------------ | Branch (307:3): [True: 23.3M, False: 0] ------------------ 308| 23.3M| data = next; 309| 23.3M| last_bits = best_bits; 310| | 311| |#if TRACE 312| | bytestats->bitg[best_bits] += best_size; 313| |#endif 314| 23.3M| } 315| | 316| |#if TRACE 317| | bytestats->header += header_size; 318| |#endif 319| | 320| 1.54M| return data; 321| 1.54M|} vertexcodec.cpp:_ZN7meshoptL16encodeBytesGroupEPhPKhi: 211| 23.3M|{ 212| 23.3M| assert(bits >= 0 && bits <= 8); ------------------ | Branch (212:2): [True: 23.3M, False: 0] | Branch (212:2): [True: 23.3M, False: 0] | Branch (212:2): [True: 23.3M, False: 0] ------------------ 213| 23.3M| assert(kByteGroupSize % 8 == 0); ------------------ | Branch (213:2): [True: 23.3M, Folded] ------------------ 214| | 215| 23.3M| if (bits == 0) ------------------ | Branch (215:6): [True: 3.47M, False: 19.8M] ------------------ 216| 3.47M| return data; 217| | 218| 19.8M| if (bits == 8) ------------------ | Branch (218:6): [True: 13.3M, False: 6.49M] ------------------ 219| 13.3M| { 220| 13.3M| memcpy(data, buffer, kByteGroupSize); 221| 13.3M| return data + kByteGroupSize; 222| 13.3M| } 223| | 224| 6.49M| size_t byte_size = 8 / bits; 225| 6.49M| assert(kByteGroupSize % byte_size == 0); ------------------ | Branch (225:2): [True: 6.49M, False: 0] ------------------ 226| | 227| | // fixed portion: bits bits for each value 228| | // variable portion: full byte for each out-of-range value (using 1...1 as sentinel) 229| 6.49M| unsigned char sentinel = (1 << bits) - 1; 230| | 231| 24.6M| for (size_t i = 0; i < kByteGroupSize; i += byte_size) ------------------ | Branch (231:21): [True: 18.2M, False: 6.49M] ------------------ 232| 18.2M| { 233| 18.2M| unsigned char byte = 0; 234| | 235| 122M| for (size_t k = 0; k < byte_size; ++k) ------------------ | Branch (235:22): [True: 103M, False: 18.2M] ------------------ 236| 103M| { 237| 103M| unsigned char enc = (buffer[i + k] >= sentinel) ? sentinel : buffer[i + k]; ------------------ | Branch (237:24): [True: 33.3M, False: 70.6M] ------------------ 238| | 239| 103M| byte <<= bits; 240| 103M| byte |= enc; 241| 103M| } 242| | 243| | // encode 1-bit groups in reverse bit order 244| | // this makes them faster to decode alongside other groups 245| 18.2M| if (bits == 1) ------------------ | Branch (245:7): [True: 7.96M, False: 10.2M] ------------------ 246| 7.96M| byte = (unsigned char)(((byte * 0x80200802ull) & 0x0884422110ull) * 0x0101010101ull >> 32); 247| | 248| 18.2M| *data++ = byte; 249| 18.2M| } 250| | 251| 110M| for (size_t i = 0; i < kByteGroupSize; ++i) ------------------ | Branch (251:21): [True: 103M, False: 6.49M] ------------------ 252| 103M| { 253| 103M| unsigned char v = buffer[i]; 254| | 255| | // branchless append of out-of-range values 256| 103M| *data = v; 257| 103M| data += v >= sentinel; 258| 103M| } 259| | 260| 6.49M| return data; 261| 6.49M|} vertexcodec.cpp:_ZN7meshoptL21decodeVertexBlockSimdEPKhS1_PhmmS2_S1_i: 1515| 176k|{ 1516| 176k| assert(vertex_count > 0 && vertex_count <= kVertexBlockMaxSize); ------------------ | Branch (1516:2): [True: 176k, False: 0] | Branch (1516:2): [True: 176k, False: 0] | Branch (1516:2): [True: 176k, False: 0] ------------------ 1517| | 1518| 176k| unsigned char buffer[kVertexBlockMaxSize * 4]; 1519| 176k| unsigned char transposed[kVertexBlockSizeBytes]; 1520| | 1521| 176k| size_t vertex_count_aligned = (vertex_count + kByteGroupSize - 1) & ~(kByteGroupSize - 1); 1522| | 1523| | // we can decode directly into the output buffer if vertex count is aligned to 16 (delta decode works 16 vertices at a time) 1524| | // this uses strided writes and also reads the last vertex once, which is bad for performance for write-combined memory so we only enable this if configured 1525| |#ifdef MESHOPTIMIZER_VERTEXCODEC_ZEROCOPY 1526| | unsigned char* target = vertex_count == vertex_count_aligned ? vertex_data : transposed; 1527| |#else 1528| 176k| unsigned char* target = transposed; 1529| 176k|#endif 1530| | 1531| 176k| size_t control_size = version == 0 ? 0 : vertex_size / 4; ------------------ | Branch (1531:24): [True: 10.6k, False: 165k] ------------------ 1532| 176k| if (size_t(data_end - data) < control_size) ------------------ | Branch (1532:6): [True: 0, False: 176k] ------------------ 1533| 0| return NULL; 1534| | 1535| 176k| const unsigned char* control = data; 1536| 176k| data += control_size; 1537| | 1538| 644k| for (size_t k = 0; k < vertex_size; k += 4) ------------------ | Branch (1538:21): [True: 468k, False: 176k] ------------------ 1539| 468k| { 1540| 468k| unsigned char ctrl_byte = version == 0 ? 0 : control[k / 4]; ------------------ | Branch (1540:29): [True: 30.6k, False: 437k] ------------------ 1541| | 1542| 2.34M| for (size_t j = 0; j < 4; ++j) ------------------ | Branch (1542:22): [True: 1.87M, False: 467k] ------------------ 1543| 1.87M| { 1544| 1.87M| int ctrl = (ctrl_byte >> (j * 2)) & 3; 1545| | 1546| 1.87M| if (ctrl == 3) ------------------ | Branch (1546:8): [True: 461k, False: 1.41M] ------------------ 1547| 461k| { 1548| | // literal encoding; safe to over-copy due to tail 1549| 461k| if (size_t(data_end - data) < vertex_count_aligned) ------------------ | Branch (1549:9): [True: 56, False: 460k] ------------------ 1550| 56| return NULL; 1551| | 1552| 460k| memcpy(buffer + j * vertex_count_aligned, data, vertex_count_aligned); 1553| 460k| data += vertex_count; 1554| 460k| } 1555| 1.41M| else if (ctrl == 2) ------------------ | Branch (1555:13): [True: 630k, False: 781k] ------------------ 1556| 630k| { 1557| | // zero encoding 1558| 630k| memset(buffer + j * vertex_count_aligned, 0, vertex_count_aligned); 1559| 630k| } 1560| 781k| else 1561| 781k| { 1562| | // for v0, headers are mapped to 0..3; for v1, headers are mapped to 4..8 1563| 781k| int hshift = version == 0 ? 0 : 4 + ctrl; ------------------ | Branch (1563:18): [True: 122k, False: 658k] ------------------ 1564| | 1565| 781k| data = decodeBytesSimd(data, data_end, buffer + j * vertex_count_aligned, vertex_count_aligned, hshift); 1566| 781k| if (!data) ------------------ | Branch (1566:9): [True: 277, False: 781k] ------------------ 1567| 277| return NULL; 1568| 781k| } 1569| 1.87M| } 1570| | 1571| 467k| int channel = version == 0 ? 0 : channels[k / 4]; ------------------ | Branch (1571:17): [True: 30.5k, False: 437k] ------------------ 1572| | 1573| 467k| switch (channel & 3) 1574| 467k| { 1575| 432k| case 0: ------------------ | Branch (1575:3): [True: 432k, False: 35.8k] ------------------ 1576| 432k| decodeDeltas4Simd<0>(buffer, target + k, vertex_count_aligned, vertex_size, last_vertex + k, 0); 1577| 432k| break; 1578| 15.5k| case 1: ------------------ | Branch (1578:3): [True: 15.5k, False: 452k] ------------------ 1579| 15.5k| decodeDeltas4Simd<1>(buffer, target + k, vertex_count_aligned, vertex_size, last_vertex + k, 0); 1580| 15.5k| break; 1581| 20.1k| case 2: ------------------ | Branch (1581:3): [True: 20.1k, False: 447k] ------------------ 1582| 20.1k| decodeDeltas4Simd<2>(buffer, target + k, vertex_count_aligned, vertex_size, last_vertex + k, (32 - (channel >> 4)) & 31); 1583| 20.1k| break; 1584| 161| default: ------------------ | Branch (1584:3): [True: 161, False: 467k] ------------------ 1585| 161| return NULL; // invalid channel type 1586| 467k| } 1587| 467k| } 1588| | 1589| 176k| if (target == transposed) ------------------ | Branch (1589:6): [True: 176k, False: 0] ------------------ 1590| 176k| memcpy(vertex_data, transposed, vertex_count * vertex_size); 1591| | 1592| 176k| memcpy(last_vertex, &target[vertex_size * (vertex_count - 1)], vertex_size); 1593| | 1594| 176k| return data; 1595| 176k|} vertexcodec.cpp:_ZN7meshoptL15decodeBytesSimdEPKhS1_Phmi: 1370| 781k|{ 1371| 781k| assert(buffer_size % kByteGroupSize == 0); ------------------ | Branch (1371:2): [True: 781k, False: 0] ------------------ 1372| 781k| assert(kByteGroupSize == 16); ------------------ | Branch (1372:2): [True: 781k, Folded] ------------------ 1373| | 1374| | // round number of groups to 4 to get number of header bytes 1375| 781k| size_t header_size = (buffer_size / kByteGroupSize + 3) / 4; 1376| 781k| if (size_t(data_end - data) < header_size) ------------------ | Branch (1376:6): [True: 17, False: 781k] ------------------ 1377| 17| return NULL; 1378| | 1379| 781k| const unsigned char* header = data; 1380| 781k| data += header_size; 1381| | 1382| 781k| size_t i = 0; 1383| | 1384| | // fast-path: process 4 groups at a time, do a shared bounds check 1385| 3.68M| for (; i + kByteGroupSize * 4 <= buffer_size && size_t(data_end - data) >= kByteGroupDecodeLimit * 4; i += kByteGroupSize * 4) ------------------ | Branch (1385:9): [True: 2.90M, False: 777k] | Branch (1385:50): [True: 2.89M, False: 3.74k] ------------------ 1386| 2.89M| { 1387| 2.89M| size_t header_offset = i / kByteGroupSize; 1388| 2.89M| unsigned char header_byte = header[header_offset / 4]; 1389| | 1390| 2.89M|#if defined(SIMD_SSE) || defined(SIMD_AVX) 1391| | // very-fast-path: for consecutive 4 groups that are all 0-bit (v0/0, v1/0/0000) or 8-bit (v0/3333, v1/1/3333), 1392| | // the branchless decoders are slower than branching over the decoding of 4 groups and issuing a few load/store ops 1393| 2.89M| if (hshift != 5 && header_byte == 0) ------------------ | Branch (1393:7): [True: 760k, False: 2.13M] | Branch (1393:22): [True: 325k, False: 434k] ------------------ 1394| 325k| { 1395| 325k| memset(buffer + i, 0, kByteGroupSize * 4); 1396| 325k| continue; 1397| 325k| } 1398| 2.57M| else if (hshift != 4 && header_byte == 255) ------------------ | Branch (1398:12): [True: 2.35M, False: 221k] | Branch (1398:27): [True: 1.55M, False: 798k] ------------------ 1399| 1.55M| { 1400| 1.55M| memcpy(buffer + i, data, kByteGroupSize * 4); 1401| 1.55M| data += kByteGroupSize * 4; 1402| 1.55M| continue; 1403| 1.55M| } 1404| 1.01M|#endif 1405| | 1406| 1.01M| data = decodeBytesGroupSimd(data, buffer + i + kByteGroupSize * 0, hshift + ((header_byte >> 0) & 3)); 1407| 1.01M| data = decodeBytesGroupSimd(data, buffer + i + kByteGroupSize * 1, hshift + ((header_byte >> 2) & 3)); 1408| 1.01M| data = decodeBytesGroupSimd(data, buffer + i + kByteGroupSize * 2, hshift + ((header_byte >> 4) & 3)); 1409| 1.01M| data = decodeBytesGroupSimd(data, buffer + i + kByteGroupSize * 3, hshift + ((header_byte >> 6) & 3)); 1410| 1.01M| } 1411| | 1412| | // slow-path: process remaining groups 1413| 890k| for (; i < buffer_size; i += kByteGroupSize) ------------------ | Branch (1413:9): [True: 109k, False: 781k] ------------------ 1414| 109k| { 1415| 109k| if (size_t(data_end - data) < kByteGroupDecodeLimit) ------------------ | Branch (1415:7): [True: 260, False: 109k] ------------------ 1416| 260| return NULL; 1417| | 1418| 109k| size_t header_offset = i / kByteGroupSize; 1419| 109k| unsigned char header_byte = header[header_offset / 4]; 1420| | 1421| 109k| data = decodeBytesGroupSimd(data, buffer + i, hshift + ((header_byte >> ((header_offset % 4) * 2)) & 3)); 1422| 109k| } 1423| | 1424| 781k| return data; 1425| 781k|} vertexcodec.cpp:_ZN7meshoptL20decodeBytesGroupSimdEPKhPhi: 831| 4.18M|{ 832| | // 0 for 1-bit, 1 for 2-bit, 2 for 4-bit, 3 for 8-bit, and 4 for 0-bit as it makes some of the uses easier 833| 4.18M| static const int hbtn[9] = {4, 1, 2, 3, 4, 0, 1, 2, 3}; 834| | 835| 4.18M| int n = hbtn[hbits]; 836| | 837| 4.18M|#ifdef SIMD_LATENCYOPT 838| 4.18M| unsigned long long data64; 839| 4.18M| memcpy(&data64, data, 8); 840| 4.18M| data64 &= data64 >> n; 841| 4.18M| data64 &= data64 >> (n >> 1); 842| | 843| | // mask out one bit per group that is set if all group bits were 1 844| 4.18M| static const unsigned long long lanes[9] = {0, 0x55555555, 0x1111111111111111ull, 0, 0, 0xffff, 0x55555555, 0x1111111111111111ull, 0}; 845| 4.18M| int datacnt = int(_mm_popcnt_u64(data64 & lanes[hbits])); 846| 4.18M|#endif 847| | 848| | // for 8-bit groups, instead of loading the bytes through 'data', we load them through 'skip' as they are easier to preserve 849| | // for 0-bit groups, the load results get discarded because mask is always 0; in both cases the shift wraps to zero 850| 4.18M| const unsigned char* skip = data + ((2 << n) & 15); 851| | 852| 4.18M| __m128i selb = _mm_loadl_epi64(reinterpret_cast(data)); 853| 4.18M| __m128i rest = _mm_loadu_si128(reinterpret_cast(skip)); 854| | 855| | // unpack 1, 2 or 4-bit values: shuffle replicates each source byte into both halves of a 16-bit lane 856| | // mulhi extracts even and odd fields into the low byte; the results are interleaved back with shift/or 857| 4.18M| __m128i selw = _mm_shuffle_epi8(selb, _mm_loadu_si128(reinterpret_cast(kDecodeBytesGroupConfig[hbits][1]))); 858| 4.18M| __m128i sel0 = _mm_mulhi_epu16(selw, _mm_loadu_si128(reinterpret_cast(kDecodeBytesGroupConfig[hbits][2]))); 859| 4.18M| __m128i sel1 = _mm_mulhi_epu16(selw, _mm_loadu_si128(reinterpret_cast(kDecodeBytesGroupConfig[hbits][3]))); 860| 4.18M| __m128i seli = _mm_or_si128(sel0, _mm_slli_epi16(sel1, 8)); 861| | 862| | // the interleaved fields are masked by the bit count (special handling: for 0/8-bit values, mul produces 0) 863| 4.18M| __m128i sent = _mm_loadu_si128(reinterpret_cast(kDecodeBytesGroupConfig[hbits][0])); 864| 4.18M| __m128i sel = _mm_and_si128(seli, sent); 865| | 866| | // compare sel to sentinel; returns 0 for 0-bit (mul produces 0, sent is 1), 1 for 8-bit (mul produces 0, sent is 0) 867| 4.18M| __m128i mask = _mm_cmpeq_epi8(sel, sent); 868| 4.18M| int mask16 = _mm_movemask_epi8(mask); 869| 4.18M| unsigned char mask0 = (unsigned char)(mask16 & 255); 870| 4.18M| unsigned char mask1 = (unsigned char)(mask16 >> 8); 871| | 872| | // decode shuffle mask from two halves; second half needs to be shifted by popcount(mask0) 873| 4.18M| __m128i sm0 = _mm_loadl_epi64(reinterpret_cast(&kDecodeBytesGroupShuffle[mask0])); 874| 4.18M| __m128i sm1 = _mm_loadl_epi64(reinterpret_cast(&kDecodeBytesGroupShuffle[mask1])); 875| | 876| | // each lane of mask is 0x00 or 0xff; sad yields 255*popcount(mask0) in low word => low byte is -popcount(mask0) 877| 4.18M| __m128i npops = _mm_sad_epu8(mask, _mm_setzero_si128()); 878| 4.18M| __m128i sm1r = _mm_sub_epi8(sm1, _mm_shuffle_epi8(npops, _mm_setzero_si128())); 879| 4.18M| __m128i shuf = _mm_unpacklo_epi64(sm0, sm1r); 880| | 881| | // expand rest via shuffle mask and combine with sel; shuffle mask zeroes out bytes that are replaced by sel 882| 4.18M| __m128i result = _mm_or_si128(_mm_shuffle_epi8(rest, shuf), _mm_andnot_si128(mask, sel)); 883| | 884| 4.18M| _mm_storeu_si128(reinterpret_cast<__m128i*>(buffer), result); 885| | 886| 4.18M|#ifdef SIMD_LATENCYOPT 887| | // datacnt is 0 for 8-bit groups so we can't use skip to advance; 0-bit groups wrap the shift to zero 888| 4.18M| return data + ((2 << n) & 31) + datacnt; 889| |#else 890| | return skip + _mm_popcnt_u32(mask16); 891| |#endif 892| 4.18M|} vertexcodec.cpp:_ZN7meshoptL17decodeDeltas4SimdILi0EEEvPKhPhmmS3_i: 1430| 432k|{ 1431| 432k|#if defined(SIMD_SSE) || defined(SIMD_AVX) 1432| 432k|#define TEMP __m128i 1433| 432k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) 1434| 432k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) 1435| 432k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) 1436| 432k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) 1437| 432k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size 1438| 432k|#endif 1439| | 1440| |#ifdef SIMD_NEON 1441| |#define TEMP uint8x8_t 1442| |#define PREP() uint8x8_t pi = vreinterpret_u8_u32(vld1_lane_u32(reinterpret_cast(last_vertex), vdup_n_u32(0), 0)) 1443| |#define LOAD(i) uint8x16_t r##i = vld1q_u8(buffer + j + i * vertex_count_aligned) 1444| |#define GRP4(i) t0 = vget_low_u8(r##i), t1 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t0), 1)), t2 = vget_high_u8(r##i), t3 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t2), 1)) 1445| |#define FIXD(i) t##i = pi = Channel == 0 ? vadd_u8(pi, t##i) : (Channel == 1 ? vreinterpret_u8_u16(vadd_u16(vreinterpret_u16_u8(pi), vreinterpret_u16_u8(t##i))) : veor_u8(pi, t##i)) 1446| |#define SAVE(i) vst1_lane_u32(reinterpret_cast(savep), vreinterpret_u32_u8(t##i), 0), savep += vertex_size 1447| |#endif 1448| | 1449| |#ifdef SIMD_WASM 1450| |#define TEMP v128_t 1451| |#define PREP() v128_t pi = wasm_v128_load(last_vertex) 1452| |#define LOAD(i) v128_t r##i = wasm_v128_load(buffer + j + i * vertex_count_aligned) 1453| |#define GRP4(i) t0 = r##i, t1 = wasmx_splat_v32x4(r##i, 1), t2 = wasmx_splat_v32x4(r##i, 2), t3 = wasmx_splat_v32x4(r##i, 3) 1454| |#define FIXD(i) t##i = pi = Channel == 0 ? wasm_i8x16_add(pi, t##i) : (Channel == 1 ? wasm_i16x8_add(pi, t##i) : wasm_v128_xor(pi, t##i)) 1455| |#define SAVE(i) wasm_v128_store32_lane(savep, t##i, 0), savep += vertex_size 1456| |#endif 1457| | 1458| 432k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) 1459| | 1460| 432k| PREP(); ------------------ | | 1433| 432k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) ------------------ 1461| | 1462| 432k| unsigned char* savep = transposed; 1463| | 1464| 6.99M| for (size_t j = 0; j < vertex_count_aligned; j += 16) ------------------ | Branch (1464:21): [True: 6.56M, False: 432k] ------------------ 1465| 6.56M| { 1466| 6.56M| LOAD(0); ------------------ | | 1434| 6.56M|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1467| 6.56M| LOAD(1); ------------------ | | 1434| 6.56M|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1468| 6.56M| LOAD(2); ------------------ | | 1434| 6.56M|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1469| 6.56M| LOAD(3); ------------------ | | 1434| 6.56M|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1470| | 1471| 6.56M| transpose8(r0, r1, r2, r3); 1472| | 1473| 6.56M| TEMP t0, t1, t2, t3; ------------------ | | 1432| 6.56M|#define TEMP __m128i ------------------ 1474| 6.56M| TEMP npi = pi; ------------------ | | 1432| 6.56M|#define TEMP __m128i ------------------ 1475| | 1476| 6.56M| UNZR(0); ------------------ | | 1458| 6.56M|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [True: 6.56M, Folded] | | | Branch (1458:58): [Folded, False: 0] | | ------------------ ------------------ 1477| 6.56M| GRP4(0); ------------------ | | 1435| 6.56M|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1478| 6.56M| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ 1479| 6.56M| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1480| | 1481| 6.56M| UNZR(1); ------------------ | | 1458| 6.56M|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [True: 6.56M, Folded] | | | Branch (1458:58): [Folded, False: 0] | | ------------------ ------------------ 1482| 6.56M| GRP4(1); ------------------ | | 1435| 6.56M|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1483| 6.56M| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ 1484| 6.56M| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1485| | 1486| 6.56M| UNZR(2); ------------------ | | 1458| 6.56M|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [True: 6.56M, Folded] | | | Branch (1458:58): [Folded, False: 0] | | ------------------ ------------------ 1487| 6.56M| GRP4(2); ------------------ | | 1435| 6.56M|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1488| 6.56M| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ 1489| 6.56M| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1490| | 1491| 6.56M| UNZR(3); ------------------ | | 1458| 6.56M|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [True: 6.56M, Folded] | | | Branch (1458:58): [Folded, False: 0] | | ------------------ ------------------ 1492| 6.56M| GRP4(3); ------------------ | | 1435| 6.56M|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1493| 6.56M| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 6.56M|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [True: 6.56M, Folded] | | | Branch (1436:70): [Folded, False: 0] | | ------------------ ------------------ 1494| 6.56M| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 6.56M|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1495| | 1496| |#if defined(SIMD_LATENCYOPT) && defined(SIMD_NEON) && (defined(__APPLE__) || defined(_WIN32)) 1497| | // instead of relying on accumulated pi, recompute it from scratch from r0..r3; this shortens dependency between loop iterations 1498| | pi = rebase(npi, r0, r1, r2, r3); 1499| |#else 1500| 6.56M| (void)npi; 1501| 6.56M|#endif 1502| | 1503| 6.56M|#undef UNZR 1504| 6.56M|#undef TEMP 1505| 6.56M|#undef PREP 1506| 6.56M|#undef LOAD 1507| 6.56M|#undef GRP4 1508| 6.56M|#undef FIXD 1509| 6.56M|#undef SAVE 1510| 6.56M| } 1511| 432k|} _ZN7meshopt10transpose8ERDv2_xS1_S1_S1_: 1219| 7.10M|{ 1220| 7.10M| __m128i t0 = _mm_unpacklo_epi8(x0, x1); 1221| 7.10M| __m128i t1 = _mm_unpackhi_epi8(x0, x1); 1222| 7.10M| __m128i t2 = _mm_unpacklo_epi8(x2, x3); 1223| 7.10M| __m128i t3 = _mm_unpackhi_epi8(x2, x3); 1224| | 1225| 7.10M| x0 = _mm_unpacklo_epi16(t0, t2); 1226| 7.10M| x1 = _mm_unpackhi_epi16(t0, t2); 1227| 7.10M| x2 = _mm_unpacklo_epi16(t1, t3); 1228| 7.10M| x3 = _mm_unpackhi_epi16(t1, t3); 1229| 7.10M|} _ZN7meshopt9unzigzag8EDv2_x: 1233| 26.2M|{ 1234| 26.2M| __m128i xl = _mm_sub_epi8(_mm_setzero_si128(), _mm_and_si128(v, _mm_set1_epi8(1))); 1235| 26.2M| __m128i xr = _mm_and_si128(_mm_srli_epi16(v, 1), _mm_set1_epi8(127)); 1236| | 1237| 26.2M| return _mm_xor_si128(xl, xr); 1238| 26.2M|} vertexcodec.cpp:_ZN7meshoptL17decodeDeltas4SimdILi1EEEvPKhPhmmS3_i: 1430| 15.5k|{ 1431| 15.5k|#if defined(SIMD_SSE) || defined(SIMD_AVX) 1432| 15.5k|#define TEMP __m128i 1433| 15.5k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) 1434| 15.5k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) 1435| 15.5k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) 1436| 15.5k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) 1437| 15.5k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size 1438| 15.5k|#endif 1439| | 1440| |#ifdef SIMD_NEON 1441| |#define TEMP uint8x8_t 1442| |#define PREP() uint8x8_t pi = vreinterpret_u8_u32(vld1_lane_u32(reinterpret_cast(last_vertex), vdup_n_u32(0), 0)) 1443| |#define LOAD(i) uint8x16_t r##i = vld1q_u8(buffer + j + i * vertex_count_aligned) 1444| |#define GRP4(i) t0 = vget_low_u8(r##i), t1 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t0), 1)), t2 = vget_high_u8(r##i), t3 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t2), 1)) 1445| |#define FIXD(i) t##i = pi = Channel == 0 ? vadd_u8(pi, t##i) : (Channel == 1 ? vreinterpret_u8_u16(vadd_u16(vreinterpret_u16_u8(pi), vreinterpret_u16_u8(t##i))) : veor_u8(pi, t##i)) 1446| |#define SAVE(i) vst1_lane_u32(reinterpret_cast(savep), vreinterpret_u32_u8(t##i), 0), savep += vertex_size 1447| |#endif 1448| | 1449| |#ifdef SIMD_WASM 1450| |#define TEMP v128_t 1451| |#define PREP() v128_t pi = wasm_v128_load(last_vertex) 1452| |#define LOAD(i) v128_t r##i = wasm_v128_load(buffer + j + i * vertex_count_aligned) 1453| |#define GRP4(i) t0 = r##i, t1 = wasmx_splat_v32x4(r##i, 1), t2 = wasmx_splat_v32x4(r##i, 2), t3 = wasmx_splat_v32x4(r##i, 3) 1454| |#define FIXD(i) t##i = pi = Channel == 0 ? wasm_i8x16_add(pi, t##i) : (Channel == 1 ? wasm_i16x8_add(pi, t##i) : wasm_v128_xor(pi, t##i)) 1455| |#define SAVE(i) wasm_v128_store32_lane(savep, t##i, 0), savep += vertex_size 1456| |#endif 1457| | 1458| 15.5k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) 1459| | 1460| 15.5k| PREP(); ------------------ | | 1433| 15.5k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) ------------------ 1461| | 1462| 15.5k| unsigned char* savep = transposed; 1463| | 1464| 243k| for (size_t j = 0; j < vertex_count_aligned; j += 16) ------------------ | Branch (1464:21): [True: 227k, False: 15.5k] ------------------ 1465| 227k| { 1466| 227k| LOAD(0); ------------------ | | 1434| 227k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1467| 227k| LOAD(1); ------------------ | | 1434| 227k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1468| 227k| LOAD(2); ------------------ | | 1434| 227k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1469| 227k| LOAD(3); ------------------ | | 1434| 227k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1470| | 1471| 227k| transpose8(r0, r1, r2, r3); 1472| | 1473| 227k| TEMP t0, t1, t2, t3; ------------------ | | 1432| 227k|#define TEMP __m128i ------------------ 1474| 227k| TEMP npi = pi; ------------------ | | 1432| 227k|#define TEMP __m128i ------------------ 1475| | 1476| 227k| UNZR(0); ------------------ | | 1458| 227k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 227k] | | | Branch (1458:58): [True: 227k, Folded] | | ------------------ ------------------ 1477| 227k| GRP4(0); ------------------ | | 1435| 227k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1478| 227k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ 1479| 227k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1480| | 1481| 227k| UNZR(1); ------------------ | | 1458| 227k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 227k] | | | Branch (1458:58): [True: 227k, Folded] | | ------------------ ------------------ 1482| 227k| GRP4(1); ------------------ | | 1435| 227k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1483| 227k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ 1484| 227k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1485| | 1486| 227k| UNZR(2); ------------------ | | 1458| 227k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 227k] | | | Branch (1458:58): [True: 227k, Folded] | | ------------------ ------------------ 1487| 227k| GRP4(2); ------------------ | | 1435| 227k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1488| 227k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ 1489| 227k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1490| | 1491| 227k| UNZR(3); ------------------ | | 1458| 227k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 227k] | | | Branch (1458:58): [True: 227k, Folded] | | ------------------ ------------------ 1492| 227k| GRP4(3); ------------------ | | 1435| 227k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1493| 227k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 227k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 227k] | | | Branch (1436:70): [True: 227k, Folded] | | ------------------ ------------------ 1494| 227k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 227k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1495| | 1496| |#if defined(SIMD_LATENCYOPT) && defined(SIMD_NEON) && (defined(__APPLE__) || defined(_WIN32)) 1497| | // instead of relying on accumulated pi, recompute it from scratch from r0..r3; this shortens dependency between loop iterations 1498| | pi = rebase(npi, r0, r1, r2, r3); 1499| |#else 1500| 227k| (void)npi; 1501| 227k|#endif 1502| | 1503| 227k|#undef UNZR 1504| 227k|#undef TEMP 1505| 227k|#undef PREP 1506| 227k|#undef LOAD 1507| 227k|#undef GRP4 1508| 227k|#undef FIXD 1509| 227k|#undef SAVE 1510| 227k| } 1511| 15.5k|} _ZN7meshopt10unzigzag16EDv2_x: 1242| 911k|{ 1243| 911k| __m128i xl = _mm_sub_epi16(_mm_setzero_si128(), _mm_and_si128(v, _mm_set1_epi16(1))); 1244| 911k| __m128i xr = _mm_srli_epi16(v, 1); 1245| | 1246| 911k| return _mm_xor_si128(xl, xr); 1247| 911k|} vertexcodec.cpp:_ZN7meshoptL17decodeDeltas4SimdILi2EEEvPKhPhmmS3_i: 1430| 20.1k|{ 1431| 20.1k|#if defined(SIMD_SSE) || defined(SIMD_AVX) 1432| 20.1k|#define TEMP __m128i 1433| 20.1k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) 1434| 20.1k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) 1435| 20.1k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) 1436| 20.1k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) 1437| 20.1k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size 1438| 20.1k|#endif 1439| | 1440| |#ifdef SIMD_NEON 1441| |#define TEMP uint8x8_t 1442| |#define PREP() uint8x8_t pi = vreinterpret_u8_u32(vld1_lane_u32(reinterpret_cast(last_vertex), vdup_n_u32(0), 0)) 1443| |#define LOAD(i) uint8x16_t r##i = vld1q_u8(buffer + j + i * vertex_count_aligned) 1444| |#define GRP4(i) t0 = vget_low_u8(r##i), t1 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t0), 1)), t2 = vget_high_u8(r##i), t3 = vreinterpret_u8_u32(vdup_lane_u32(vreinterpret_u32_u8(t2), 1)) 1445| |#define FIXD(i) t##i = pi = Channel == 0 ? vadd_u8(pi, t##i) : (Channel == 1 ? vreinterpret_u8_u16(vadd_u16(vreinterpret_u16_u8(pi), vreinterpret_u16_u8(t##i))) : veor_u8(pi, t##i)) 1446| |#define SAVE(i) vst1_lane_u32(reinterpret_cast(savep), vreinterpret_u32_u8(t##i), 0), savep += vertex_size 1447| |#endif 1448| | 1449| |#ifdef SIMD_WASM 1450| |#define TEMP v128_t 1451| |#define PREP() v128_t pi = wasm_v128_load(last_vertex) 1452| |#define LOAD(i) v128_t r##i = wasm_v128_load(buffer + j + i * vertex_count_aligned) 1453| |#define GRP4(i) t0 = r##i, t1 = wasmx_splat_v32x4(r##i, 1), t2 = wasmx_splat_v32x4(r##i, 2), t3 = wasmx_splat_v32x4(r##i, 3) 1454| |#define FIXD(i) t##i = pi = Channel == 0 ? wasm_i8x16_add(pi, t##i) : (Channel == 1 ? wasm_i16x8_add(pi, t##i) : wasm_v128_xor(pi, t##i)) 1455| |#define SAVE(i) wasm_v128_store32_lane(savep, t##i, 0), savep += vertex_size 1456| |#endif 1457| | 1458| 20.1k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) 1459| | 1460| 20.1k| PREP(); ------------------ | | 1433| 20.1k|#define PREP() __m128i pi = _mm_cvtsi32_si128(*reinterpret_cast(last_vertex)) ------------------ 1461| | 1462| 20.1k| unsigned char* savep = transposed; 1463| | 1464| 326k| for (size_t j = 0; j < vertex_count_aligned; j += 16) ------------------ | Branch (1464:21): [True: 306k, False: 20.1k] ------------------ 1465| 306k| { 1466| 306k| LOAD(0); ------------------ | | 1434| 306k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1467| 306k| LOAD(1); ------------------ | | 1434| 306k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1468| 306k| LOAD(2); ------------------ | | 1434| 306k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1469| 306k| LOAD(3); ------------------ | | 1434| 306k|#define LOAD(i) __m128i r##i = _mm_loadu_si128(reinterpret_cast(buffer + j + i * vertex_count_aligned)) ------------------ 1470| | 1471| 306k| transpose8(r0, r1, r2, r3); 1472| | 1473| 306k| TEMP t0, t1, t2, t3; ------------------ | | 1432| 306k|#define TEMP __m128i ------------------ 1474| 306k| TEMP npi = pi; ------------------ | | 1432| 306k|#define TEMP __m128i ------------------ 1475| | 1476| 306k| UNZR(0); ------------------ | | 1458| 306k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 306k] | | | Branch (1458:58): [Folded, False: 306k] | | ------------------ ------------------ 1477| 306k| GRP4(0); ------------------ | | 1435| 306k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1478| 306k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ 1479| 306k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1480| | 1481| 306k| UNZR(1); ------------------ | | 1458| 306k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 306k] | | | Branch (1458:58): [Folded, False: 306k] | | ------------------ ------------------ 1482| 306k| GRP4(1); ------------------ | | 1435| 306k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1483| 306k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ 1484| 306k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1485| | 1486| 306k| UNZR(2); ------------------ | | 1458| 306k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 306k] | | | Branch (1458:58): [Folded, False: 306k] | | ------------------ ------------------ 1487| 306k| GRP4(2); ------------------ | | 1435| 306k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1488| 306k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ 1489| 306k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1490| | 1491| 306k| UNZR(3); ------------------ | | 1458| 306k|#define UNZR(i) r##i = Channel == 0 ? unzigzag8(r##i) : (Channel == 1 ? unzigzag16(r##i) : rotate32(r##i, rot)) | | ------------------ | | | Branch (1458:24): [Folded, False: 306k] | | | Branch (1458:58): [Folded, False: 306k] | | ------------------ ------------------ 1492| 306k| GRP4(3); ------------------ | | 1435| 306k|#define GRP4(i) t0 = r##i, t1 = _mm_shuffle_epi32(r##i, 1), t2 = _mm_shuffle_epi32(r##i, 2), t3 = _mm_shuffle_epi32(r##i, 3) ------------------ 1493| 306k| FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ FIXD(0), FIXD(1), FIXD(2), FIXD(3); ------------------ | | 1436| 306k|#define FIXD(i) t##i = pi = Channel == 0 ? _mm_add_epi8(pi, t##i) : (Channel == 1 ? _mm_add_epi16(pi, t##i) : _mm_xor_si128(pi, t##i)) | | ------------------ | | | Branch (1436:29): [Folded, False: 306k] | | | Branch (1436:70): [Folded, False: 306k] | | ------------------ ------------------ 1494| 306k| SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ SAVE(0), SAVE(1), SAVE(2), SAVE(3); ------------------ | | 1437| 306k|#define SAVE(i) *reinterpret_cast(savep) = _mm_cvtsi128_si32(t##i), savep += vertex_size ------------------ 1495| | 1496| |#if defined(SIMD_LATENCYOPT) && defined(SIMD_NEON) && (defined(__APPLE__) || defined(_WIN32)) 1497| | // instead of relying on accumulated pi, recompute it from scratch from r0..r3; this shortens dependency between loop iterations 1498| | pi = rebase(npi, r0, r1, r2, r3); 1499| |#else 1500| 306k| (void)npi; 1501| 306k|#endif 1502| | 1503| 306k|#undef UNZR 1504| 306k|#undef TEMP 1505| 306k|#undef PREP 1506| 306k|#undef LOAD 1507| 306k|#undef GRP4 1508| 306k|#undef FIXD 1509| 306k|#undef SAVE 1510| 306k| } 1511| 20.1k|} _ZN7meshopt8rotate32EDv2_xi: 1251| 1.22M|{ 1252| 1.22M| return _mm_or_si128(_mm_slli_epi32(v, r), _mm_srli_epi32(v, 32 - r)); 1253| 1.22M|} _Z11fuzzDecoderPKhmmPFiPvmmS0_mE: 8| 17.2k|{ 9| 17.2k| size_t count = 66; // must be divisible by 3 for decodeIndexBuffer; should be >=64 to cover large vertex blocks 10| | 11| 17.2k| void* destination = malloc(count * stride); 12| 17.2k| assert(destination); ------------------ | Branch (12:2): [True: 17.2k, False: 0] ------------------ 13| | 14| 17.2k| int rc = decode(destination, count, stride, reinterpret_cast(data), size); 15| 17.2k| (void)rc; 16| | 17| 17.2k| free(destination); 18| 17.2k|} _Z13fuzzRoundtripPKhmmi: 21| 8.62k|{ 22| 8.62k| size_t count = size / stride; 23| | 24| 8.62k| size_t bound = meshopt_encodeVertexBufferBound(count, stride); 25| 8.62k| void* encoded = malloc(bound); 26| 8.62k| void* decoded = malloc(count * stride); 27| 8.62k| assert(encoded && decoded); ------------------ | Branch (27:2): [True: 8.62k, False: 0] | Branch (27:2): [True: 8.62k, False: 0] | Branch (27:2): [True: 8.62k, False: 0] ------------------ 28| | 29| 8.62k| size_t res = meshopt_encodeVertexBufferLevel(static_cast(encoded), bound, data, count, stride, level, -1); 30| 8.62k| assert(res > 0 && res <= bound); ------------------ | Branch (30:2): [True: 8.62k, False: 0] | Branch (30:2): [True: 8.62k, False: 0] | Branch (30:2): [True: 8.62k, False: 0] ------------------ 31| | 32| | // encode again at the boundary to check for memory safety 33| | // this should produce the same output because encoder is deterministic 34| 8.62k| size_t rese = meshopt_encodeVertexBufferLevel(static_cast(encoded) + bound - res, res, data, count, stride, level, -1); 35| 8.62k| assert(rese == res); ------------------ | Branch (35:2): [True: 8.62k, False: 0] ------------------ 36| | 37| 8.62k| int rc = meshopt_decodeVertexBuffer(decoded, count, stride, static_cast(encoded) + bound - res, res); 38| 8.62k| assert(rc == 0); ------------------ | Branch (38:2): [True: 8.62k, False: 0] ------------------ 39| | 40| 8.62k| assert(memcmp(data, decoded, count * stride) == 0); ------------------ | Branch (40:2): [True: 8.62k, False: 0] ------------------ 41| | 42| 8.62k| free(decoded); 43| 8.62k| free(encoded); 44| 8.62k|} _Z5alignmm: 47| 12.8k|{ 48| 12.8k| return (value + alignment - 1) & ~(alignment - 1); 49| 12.8k|} _Z17fuzzDecodeMeshletmmPKhm: 52| 2.14k|{ 53| | // raw decoding: allowed to write align(count, 4) elements 54| 2.14k| unsigned int rt[256]; 55| 2.14k| unsigned int rv[256]; 56| 2.14k| meshopt_decodeMeshletRaw(rv + 256 - align(vertex_count, 4), vertex_count, rt + 256 - align(triangle_count, 4), triangle_count, data, size); 57| | 58| | // regular decoding: allowed to write align(count * size, 4) bytes 59| | // with variations for 3-byte triangles and 2-byte vertex references 60| 2.14k| unsigned short rsv[256]; 61| 2.14k| unsigned char rbt[256 * 3]; 62| | 63| 2.14k| meshopt_decodeMeshlet(rv + 256 - vertex_count, vertex_count, 4, rt + 256 - triangle_count, triangle_count, 4, data, size); 64| 2.14k| meshopt_decodeMeshlet(rsv + 256 - align(vertex_count, 2), vertex_count, 2, rt + 256 - triangle_count, triangle_count, 4, data, size); 65| 2.14k| meshopt_decodeMeshlet(rv + 256 - vertex_count, vertex_count, 4, rbt + 256 * 3 - align(triangle_count * 3, 4), triangle_count, 3, data, size); 66| 2.14k| meshopt_decodeMeshlet(rsv + 256 - align(vertex_count, 2), vertex_count, 2, rbt + 256 * 3 - align(triangle_count * 3, 4), triangle_count, 3, data, size); 67| 2.14k|} _Z20fuzzRoundtripMeshletPKhm: 70| 2.15k|{ 71| 2.15k| size_t triangle_count = size / 3; 72| 2.15k| if (triangle_count > 256) ------------------ | Branch (72:6): [True: 530, False: 1.62k] ------------------ 73| 530| triangle_count = 256; 74| | 75| 2.15k| unsigned char buf[4096]; 76| 2.15k| size_t enc = meshopt_encodeMeshlet(buf, sizeof(buf), NULL, 0, reinterpret_cast(data), triangle_count); 77| 2.15k| assert(enc > 0); ------------------ | Branch (77:2): [True: 2.15k, False: 0] ------------------ 78| 2.15k| assert(enc <= meshopt_encodeMeshletBound(0, triangle_count)); ------------------ | Branch (78:2): [True: 2.15k, False: 0] ------------------ 79| | 80| 2.15k| unsigned int rt4[256]; 81| 2.15k| int rc4 = meshopt_decodeMeshlet(static_cast(NULL), 0, rt4, triangle_count, buf, enc); 82| 2.15k| assert(rc4 == 0); ------------------ | Branch (82:2): [True: 2.15k, False: 0] ------------------ 83| | 84| 180k| for (size_t i = 0; i < triangle_count; ++i) ------------------ | Branch (84:21): [True: 177k, False: 2.15k] ------------------ 85| 177k| { 86| 177k| unsigned char a = data[i * 3 + 0], b = data[i * 3 + 1], c = data[i * 3 + 2]; 87| | 88| 177k| unsigned int abc = (a << 0) | (b << 8) | (c << 16); 89| 177k| unsigned int bca = (b << 0) | (c << 8) | (a << 16); 90| 177k| unsigned int cba = (c << 0) | (a << 8) | (b << 16); 91| | 92| 177k| unsigned int tri = rt4[i]; 93| | 94| 177k| assert(tri == abc || tri == bca || tri == cba); ------------------ | Branch (94:3): [True: 107k, False: 70.1k] | Branch (94:3): [True: 35.7k, False: 34.3k] | Branch (94:3): [True: 34.3k, False: 0] | Branch (94:3): [True: 177k, False: 0] ------------------ 95| 177k| } 96| | 97| 2.15k| unsigned char rt3[256 * 3]; 98| 2.15k| int rc3 = meshopt_decodeMeshlet(static_cast(NULL), 0, rt3, triangle_count, buf, enc); 99| 2.15k| assert(rc3 == 0); ------------------ | Branch (99:2): [True: 2.15k, False: 0] ------------------ 100| | 101| 180k| for (size_t i = 0; i < triangle_count; ++i) ------------------ | Branch (101:21): [True: 177k, False: 2.15k] ------------------ 102| 177k| { 103| 177k| unsigned char a = data[i * 3 + 0], b = data[i * 3 + 1], c = data[i * 3 + 2]; 104| | 105| 177k| unsigned int abc = (a << 0) | (b << 8) | (c << 16); 106| 177k| unsigned int bca = (b << 0) | (c << 8) | (a << 16); 107| 177k| unsigned int cba = (c << 0) | (a << 8) | (b << 16); 108| | 109| 177k| unsigned int tri = rt3[i * 3 + 0] | (rt3[i * 3 + 1] << 8) | (rt3[i * 3 + 2] << 16); 110| | 111| | assert(tri == abc || tri == bca || tri == cba); ------------------ | Branch (111:3): [True: 107k, False: 70.1k] | Branch (111:3): [True: 35.7k, False: 34.3k] | Branch (111:3): [True: 34.3k, False: 0] | Branch (111:3): [True: 177k, False: 0] ------------------ 112| 177k| } 113| 2.15k|} _Z21fuzzRoundtripMeshletVPKhm: 116| 2.15k|{ 117| 2.15k| size_t vertex_count = size / 4; 118| 2.15k| if (vertex_count > 256) ------------------ | Branch (118:6): [True: 477, False: 1.68k] ------------------ 119| 477| vertex_count = 256; 120| | 121| 2.15k| unsigned char tri[4] = {0, 1, 2}; 122| | 123| 2.15k| unsigned char buf[4096]; 124| 2.15k| size_t enc = meshopt_encodeMeshlet(buf, sizeof(buf), reinterpret_cast(data), vertex_count, tri, 1); 125| 2.15k| assert(enc > 0); ------------------ | Branch (125:2): [True: 2.15k, False: 0] ------------------ 126| 2.15k| assert(enc <= meshopt_encodeMeshletBound(vertex_count, 1)); ------------------ | Branch (126:2): [True: 2.15k, False: 0] ------------------ 127| | 128| 2.15k| unsigned int rv4[256]; 129| 2.15k| int rc4 = meshopt_decodeMeshlet(rv4, vertex_count, tri, 1, buf, enc); 130| 2.15k| assert(rc4 == 0); ------------------ | Branch (130:2): [True: 2.15k, False: 0] ------------------ 131| | 132| 168k| for (size_t i = 0; i < vertex_count; ++i) ------------------ | Branch (132:21): [True: 166k, False: 2.15k] ------------------ 133| 166k| assert(rv4[i] == reinterpret_cast(data)[i]); ------------------ | Branch (133:3): [True: 166k, False: 0] ------------------ 134| | 135| 2.15k| unsigned short rv2[256]; 136| 2.15k| int rc2 = meshopt_decodeMeshlet(rv2, vertex_count, tri, 1, buf, enc); 137| 2.15k| assert(rc2 == 0); ------------------ | Branch (137:2): [True: 2.15k, False: 0] ------------------ 138| | 139| 168k| for (size_t i = 0; i < vertex_count; ++i) ------------------ | Branch (139:21): [True: 166k, False: 2.15k] ------------------ 140| | assert(rv2[i] == uint16_t(reinterpret_cast(data)[i])); ------------------ | Branch (140:3): [True: 166k, False: 0] ------------------ 141| 2.15k|} LLVMFuzzerTestOneInput: 144| 2.15k|{ 145| | // decodeIndexBuffer supports 2 and 4-byte indices 146| 2.15k| fuzzDecoder(data, size, 2, meshopt_decodeIndexBuffer); 147| 2.15k| fuzzDecoder(data, size, 4, meshopt_decodeIndexBuffer); 148| | 149| | // decodeIndexSequence supports 2 and 4-byte indices 150| 2.15k| fuzzDecoder(data, size, 2, meshopt_decodeIndexSequence); 151| 2.15k| fuzzDecoder(data, size, 4, meshopt_decodeIndexSequence); 152| | 153| | // decodeVertexBuffer supports any strides divisible by 4 in 4-256 interval 154| | // it's a waste of time to check all of them, so we'll just check a few with different alignment mod 16 155| 2.15k| fuzzDecoder(data, size, 4, meshopt_decodeVertexBuffer); 156| 2.15k| fuzzDecoder(data, size, 16, meshopt_decodeVertexBuffer); 157| 2.15k| fuzzDecoder(data, size, 24, meshopt_decodeVertexBuffer); 158| 2.15k| fuzzDecoder(data, size, 32, meshopt_decodeVertexBuffer); 159| | 160| | // encodeVertexBuffer/decodeVertexBuffer should roundtrip for any stride, check a few with different alignment mod 16 161| | // this also checks memory safety properties of the encoder 162| | // to conserve time, we only check one version/level combination, biased towards version 1 163| 2.15k| uint8_t data0 = size > 0 ? data[0] : 0; ------------------ | Branch (163:18): [True: 2.15k, False: 0] ------------------ 164| 2.15k| int level = data0 % 5; 165| | 166| 2.15k| meshopt_encodeVertexVersion(level < 4 ? 1 : 0); ------------------ | Branch (166:30): [True: 1.68k, False: 477] ------------------ 167| | 168| 2.15k| fuzzRoundtrip(data, size, 4, level); 169| 2.15k| fuzzRoundtrip(data, size, 16, level); 170| 2.15k| fuzzRoundtrip(data, size, 24, level); 171| 2.15k| fuzzRoundtrip(data, size, 32, level); 172| | 173| | // validate that decodeMeshlet works on untrusted data and is memory safe within documented limits 174| 2.15k| if (size > 2) ------------------ | Branch (174:6): [True: 2.14k, False: 14] ------------------ 175| 2.14k| fuzzDecodeMeshlet(data[0] + 1, data[1] + 1, reinterpret_cast(data + 2), size - 2); 176| | 177| | // validate that index data roundtrips in meshlet encoding modulo rotation 178| 2.15k| fuzzRoundtripMeshlet(data, size); 179| | 180| | // validate that vertex data roundtrips in meshlet encoding 181| 2.15k| fuzzRoundtripMeshletV(data, size); 182| | 183| 2.15k| return 0; 184| 2.15k|}