/src/build-dir/_deps/simdjson-src/src/haswell.cpp
Line | Count | Source |
1 | | #ifndef SIMDJSON_SRC_HASWELL_CPP |
2 | | #define SIMDJSON_SRC_HASWELL_CPP |
3 | | |
4 | | #ifndef SIMDJSON_CONDITIONAL_INCLUDE |
5 | | #include <base.h> |
6 | | #endif // SIMDJSON_CONDITIONAL_INCLUDE |
7 | | |
8 | | #include <simdjson/haswell.h> |
9 | | #include <simdjson/haswell/implementation.h> |
10 | | |
11 | | #include <simdjson/haswell/begin.h> |
12 | | #include <generic/amalgamated.h> |
13 | | #include <generic/stage1/amalgamated.h> |
14 | | #include <generic/stage2/amalgamated.h> |
15 | | |
16 | | // |
17 | | // Stage 1 |
18 | | // |
19 | | |
20 | | namespace simdjson { |
21 | | namespace SIMDJSON_IMPLEMENTATION { |
22 | | |
23 | | simdjson_warn_unused error_code implementation::create_dom_parser_implementation( |
24 | | size_t capacity, |
25 | | size_t max_depth, |
26 | | std::unique_ptr<internal::dom_parser_implementation>& dst |
27 | 0 | ) const noexcept { |
28 | 0 | dst.reset( new (std::nothrow) dom_parser_implementation() ); |
29 | 0 | if (!dst) { return MEMALLOC; } |
30 | 0 | if (auto err = dst->set_capacity(capacity)) |
31 | 0 | return err; |
32 | 0 | if (auto err = dst->set_max_depth(max_depth)) |
33 | 0 | return err; |
34 | 0 | return SUCCESS; |
35 | 0 | } |
36 | | |
37 | | namespace { |
38 | | |
39 | | using namespace simd; |
40 | | |
41 | | // This identifies structural characters (comma, colon, braces, brackets), |
42 | | // and ASCII white-space ('\r','\n','\t',' '). |
43 | 0 | simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) { |
44 | | // These lookups rely on the fact that anything < 127 will match the lower 4 bits, which is why |
45 | | // we can't use the generic lookup_16. |
46 | 0 | const auto whitespace_table = simd8<uint8_t>::repeat_16(' ', 100, 100, 100, 17, 100, 113, 2, 100, '\t', '\n', 112, 100, '\r', 100, 100); |
47 | | |
48 | | // The 6 operators (:,[]{}) have these values: |
49 | | // |
50 | | // , 2C |
51 | | // : 3A |
52 | | // [ 5B |
53 | | // { 7B |
54 | | // ] 5D |
55 | | // } 7D |
56 | | // |
57 | | // If you use | 0x20 to turn [ and ] into { and }, the lower 4 bits of each character is unique. |
58 | | // We exploit this, using a simd 4-bit lookup to tell us which character match against, and then |
59 | | // match it (against | 0x20). |
60 | | // |
61 | | // To prevent recognizing other characters, everything else gets compared with 0, which cannot |
62 | | // match due to the | 0x20. |
63 | | // |
64 | | // NOTE: Due to the | 0x20, this ALSO treats <FF> and <SUB> (control characters 0C and 1A) like , |
65 | | // and :. This gets caught in stage 2, which checks the actual character to ensure the right |
66 | | // operators are in the right places. |
67 | 0 | const auto op_table = simd8<uint8_t>::repeat_16( |
68 | 0 | 0, 0, 0, 0, |
69 | 0 | 0, 0, 0, 0, |
70 | 0 | 0, 0, ':', '{', // : = 3A, [ = 5B, { = 7B |
71 | 0 | ',', '}', 0, 0 // , = 2C, ] = 5D, } = 7D |
72 | 0 | ); |
73 | | |
74 | | // We compute whitespace and op separately. If later code only uses one or the |
75 | | // other, given the fact that all functions are aggressively inlined, we can |
76 | | // hope that useless computations will be omitted. This is namely case when |
77 | | // minifying (we only need whitespace). |
78 | |
|
79 | 0 | const uint64_t whitespace = in.eq({ |
80 | 0 | _mm256_shuffle_epi8(whitespace_table, in.chunks[0]), |
81 | 0 | _mm256_shuffle_epi8(whitespace_table, in.chunks[1]) |
82 | 0 | }); |
83 | | // Turn [ and ] into { and } |
84 | 0 | const simd8x64<uint8_t> curlified{ |
85 | 0 | in.chunks[0] | 0x20, |
86 | 0 | in.chunks[1] | 0x20 |
87 | 0 | }; |
88 | 0 | const uint64_t op = curlified.eq({ |
89 | 0 | _mm256_shuffle_epi8(op_table, in.chunks[0]), |
90 | 0 | _mm256_shuffle_epi8(op_table, in.chunks[1]) |
91 | 0 | }); |
92 | |
|
93 | 0 | return { whitespace, op }; |
94 | 0 | } |
95 | | |
96 | 0 | simdjson_inline bool is_ascii(const simd8x64<uint8_t>& input) { |
97 | 0 | return input.reduce_or().is_ascii(); |
98 | 0 | } |
99 | | |
100 | 0 | simdjson_unused simdjson_inline simd8<bool> must_be_continuation(const simd8<uint8_t> prev1, const simd8<uint8_t> prev2, const simd8<uint8_t> prev3) { |
101 | 0 | simd8<uint8_t> is_second_byte = prev1.saturating_sub(0xc0u-1); // Only 11______ will be > 0 |
102 | 0 | simd8<uint8_t> is_third_byte = prev2.saturating_sub(0xe0u-1); // Only 111_____ will be > 0 |
103 | 0 | simd8<uint8_t> is_fourth_byte = prev3.saturating_sub(0xf0u-1); // Only 1111____ will be > 0 |
104 | 0 | // Caller requires a bool (all 1's). All values resulting from the subtraction will be <= 64, so signed comparison is fine. |
105 | 0 | return simd8<int8_t>(is_second_byte | is_third_byte | is_fourth_byte) > int8_t(0); |
106 | 0 | } |
107 | | |
108 | 0 | simdjson_inline simd8<uint8_t> must_be_2_3_continuation(const simd8<uint8_t> prev2, const simd8<uint8_t> prev3) { |
109 | 0 | simd8<uint8_t> is_third_byte = prev2.saturating_sub(0xe0u-0x80); // Only 111_____ will be >= 0x80 |
110 | 0 | simd8<uint8_t> is_fourth_byte = prev3.saturating_sub(0xf0u-0x80); // Only 1111____ will be >= 0x80 |
111 | 0 | return is_third_byte | is_fourth_byte; |
112 | 0 | } |
113 | | |
114 | | } // unnamed namespace |
115 | | } // namespace SIMDJSON_IMPLEMENTATION |
116 | | } // namespace simdjson |
117 | | |
118 | | // |
119 | | // Stage 2 |
120 | | // |
121 | | |
122 | | // |
123 | | // Implementation-specific overrides |
124 | | // |
125 | | namespace simdjson { |
126 | | namespace SIMDJSON_IMPLEMENTATION { |
127 | | |
128 | 0 | simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept { |
129 | 0 | return haswell::stage1::json_minifier::minify<128>(buf, len, dst, dst_len); |
130 | 0 | } |
131 | | |
132 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept { |
133 | 0 | this->buf = _buf; |
134 | 0 | this->len = _len; |
135 | 0 | return haswell::stage1::json_structural_indexer::index<128>(_buf, _len, *this, streaming); |
136 | 0 | } |
137 | | |
138 | 0 | simdjson_warn_unused bool implementation::validate_utf8(const char *buf, size_t len) const noexcept { |
139 | 0 | return haswell::stage1::generic_validate_utf8(buf,len); |
140 | 0 | } |
141 | | |
142 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage2(dom::document &_doc) noexcept { |
143 | 0 | return stage2::tape_builder::parse_document<false>(*this, _doc); |
144 | 0 | } |
145 | | |
146 | 0 | simdjson_warn_unused error_code dom_parser_implementation::stage2_next(dom::document &_doc) noexcept { |
147 | 0 | return stage2::tape_builder::parse_document<true>(*this, _doc); |
148 | 0 | } |
149 | | |
150 | | SIMDJSON_NO_SANITIZE_MEMORY |
151 | 0 | simdjson_warn_unused uint8_t *dom_parser_implementation::parse_string(const uint8_t *src, uint8_t *dst, bool replacement_char) const noexcept { |
152 | 0 | return haswell::stringparsing::parse_string(src, dst, replacement_char); |
153 | 0 | } |
154 | | |
155 | 0 | simdjson_warn_unused uint8_t *dom_parser_implementation::parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept { |
156 | 0 | return haswell::stringparsing::parse_wobbly_string(src, dst); |
157 | 0 | } |
158 | | |
159 | 0 | simdjson_warn_unused error_code dom_parser_implementation::parse(const uint8_t *_buf, size_t _len, dom::document &_doc) noexcept { |
160 | 0 | auto error = stage1(_buf, _len, stage1_mode::regular); |
161 | 0 | if (error) { return error; } |
162 | 0 | return stage2(_doc); |
163 | 0 | } |
164 | | |
165 | | } // namespace SIMDJSON_IMPLEMENTATION |
166 | | } // namespace simdjson |
167 | | |
168 | | #include <simdjson/haswell/end.h> |
169 | | |
170 | | #endif // SIMDJSON_SRC_HASWELL_CPP |