/src/suricata8/rust/htp/src/util.rs
Line | Count | Source |
1 | | //! Utility functions for http parsing. |
2 | | |
3 | | use crate::{config::HtpServerPersonality, error::NomError}; |
4 | | use nom::{ |
5 | | branch::alt, |
6 | | bytes::complete::{ |
7 | | is_not, tag, tag_no_case, take_till, take_until, take_while, take_while1, take_while_m_n, |
8 | | }, |
9 | | bytes::streaming::{tag as streaming_tag, take_till as streaming_take_till}, |
10 | | character::complete::{char, digit1}, |
11 | | character::is_space as nom_is_space, |
12 | | combinator::{map, opt}, |
13 | | sequence::tuple, |
14 | | Err::Incomplete, |
15 | | IResult, Needed, |
16 | | }; |
17 | | |
18 | | use std::str::FromStr; |
19 | | |
20 | | /// String for the libhtp version. |
21 | | pub const HTP_VERSION_STRING_FULL: &'_ str = concat!("LibHTP v", env!("CARGO_PKG_VERSION"), "\x00"); |
22 | | |
23 | | /// Trait to allow interacting with flags. |
24 | | pub(crate) trait FlagOperations<T> { |
25 | | /// Inserts the specified flags in-place. |
26 | | fn set(&mut self, other: T); |
27 | | /// Removes the specified flags in-place. |
28 | | fn unset(&mut self, other: T); |
29 | | /// Determine if the specified flags are set |
30 | | fn is_set(&self, other: T) -> bool; |
31 | | } |
32 | | |
33 | | impl FlagOperations<u8> for u8 { |
34 | | /// Inserts the specified flags in-place. |
35 | 589k | fn set(&mut self, other: u8) { |
36 | 589k | *self |= other; |
37 | 589k | } |
38 | | /// Removes the specified flags in-place. |
39 | 0 | fn unset(&mut self, other: u8) { |
40 | 0 | *self &= !other; |
41 | 0 | } |
42 | | /// Determine if the specified flags are set |
43 | 0 | fn is_set(&self, other: u8) -> bool { |
44 | 0 | self & other != 0 |
45 | 0 | } |
46 | | } |
47 | | |
48 | | impl FlagOperations<u64> for u64 { |
49 | | /// Inserts the specified flags in-place. |
50 | 39.0M | fn set(&mut self, other: u64) { |
51 | 39.0M | *self |= other; |
52 | 39.0M | } |
53 | | /// Removes the specified flags in-place. |
54 | 31.9k | fn unset(&mut self, other: u64) { |
55 | 31.9k | *self &= !other; |
56 | 31.9k | } |
57 | | /// Determine if the specified flags are set |
58 | 24.2M | fn is_set(&self, other: u64) -> bool { |
59 | 24.2M | self & other != 0 |
60 | 24.2M | } |
61 | | } |
62 | | |
63 | | /// Various flag bits. Even though we have a flag field in several places |
64 | | /// (header, transaction, connection), these fields are all in the same namespace |
65 | | /// because we may want to set the same flag in several locations. For example, we |
66 | | /// may set HTP_FIELD_FOLDED on the actual folded header, but also on the transaction |
67 | | /// that contains the header. Both uses are useful. |
68 | | #[repr(C)] |
69 | | pub struct HtpFlags; |
70 | | |
71 | | impl HtpFlags { |
72 | | /// Field cannot be parsed. |
73 | | pub const FIELD_UNPARSEABLE: u64 = 0x0000_0000_0004; |
74 | | /// Field is invalid. |
75 | | pub const FIELD_INVALID: u64 = 0x0000_0000_0008; |
76 | | /// Field is folded. |
77 | | pub const FIELD_FOLDED: u64 = 0x0000_0000_0010; |
78 | | /// Field has been seen more than once. |
79 | | pub const FIELD_REPEATED: u64 = 0x0000_0000_0020; |
80 | | // Field is too long. |
81 | | //pub const FIELD_LONG: u64 = 0x0000_0000_0040; |
82 | | // Field contains raw null byte. |
83 | | //pub const FIELD_RAW_NUL: u64 = 0x0000_0000_0080; |
84 | | /// Detect HTTP request smuggling. |
85 | | pub const REQUEST_SMUGGLING: u64 = 0x0000_0000_0100; |
86 | | /// Invalid header folding. |
87 | | pub const INVALID_FOLDING: u64 = 0x0000_0000_0200; |
88 | | /// Invalid request transfer-encoding. |
89 | | pub const REQUEST_INVALID_T_E: u64 = 0x0000_0000_0400; |
90 | | /// Multiple chunks. |
91 | | pub const MULTI_PACKET_HEAD: u64 = 0x0000_0000_0800; |
92 | | /// No host information in header. |
93 | | pub const HOST_MISSING: u64 = 0x0000_0000_1000; |
94 | | /// Inconsistent host or port information. |
95 | | pub const HOST_AMBIGUOUS: u64 = 0x0000_0000_2000; |
96 | | /// Encoded path contains null. |
97 | | pub const PATH_ENCODED_NUL: u64 = 0x0000_0000_4000; |
98 | | /// Url encoded contains raw null. |
99 | | pub const PATH_RAW_NUL: u64 = 0x0000_0000_8000; |
100 | | /// Url encoding is invalid. |
101 | | pub const PATH_INVALID_ENCODING: u64 = 0x0000_0001_0000; |
102 | | // Path is invalid. |
103 | | //pub const PATH_INVALID: u64 = 0x0000_0002_0000; |
104 | | /// Overlong usage in path. |
105 | | pub const PATH_OVERLONG_U: u64 = 0x0000_0004_0000; |
106 | | /// Encoded path separators present. |
107 | | pub const PATH_ENCODED_SEPARATOR: u64 = 0x0000_0008_0000; |
108 | | /// At least one valid UTF-8 character and no invalid ones. |
109 | | pub const PATH_UTF8_VALID: u64 = 0x0000_0010_0000; |
110 | | /// Invalid utf8 in path. |
111 | | pub const PATH_UTF8_INVALID: u64 = 0x0000_0020_0000; |
112 | | /// Invalid utf8 overlong character. |
113 | | pub const PATH_UTF8_OVERLONG: u64 = 0x0000_0040_0000; |
114 | | /// Range U+FF00 - U+FFEF detected. |
115 | | pub const PATH_HALF_FULL_RANGE: u64 = 0x0000_0080_0000; |
116 | | /// Status line is invalid. |
117 | | pub const STATUS_LINE_INVALID: u64 = 0x0000_0100_0000; |
118 | | /// Host in the URI. |
119 | | pub const HOSTU_INVALID: u64 = 0x0000_0200_0000; |
120 | | /// Host in the Host header. |
121 | | pub const HOSTH_INVALID: u64 = 0x0000_0400_0000; |
122 | | /// Contains null. |
123 | | pub const URLEN_ENCODED_NUL: u64 = 0x0000_0800_0000; |
124 | | /// Invalid encoding. |
125 | | pub const URLEN_INVALID_ENCODING: u64 = 0x0000_1000_0000; |
126 | | /// Overlong usage. |
127 | | pub const URLEN_OVERLONG_U: u64 = 0x0000_2000_0000; |
128 | | /// Range U+FF00 - U+FFEF detected. |
129 | | pub const URLEN_HALF_FULL_RANGE: u64 = 0x0000_4000_0000; |
130 | | /// Raw null byte. |
131 | | pub const URLEN_RAW_NUL: u64 = 0x0000_8000_0000; |
132 | | /// Request invalid. |
133 | | pub const REQUEST_INVALID: u64 = 0x0001_0000_0000; |
134 | | /// Request content-length invalid. |
135 | | pub const REQUEST_INVALID_C_L: u64 = 0x0002_0000_0000; |
136 | | /// Authorization is invalid. |
137 | | pub const AUTH_INVALID: u64 = 0x0004_0000_0000; |
138 | | /// Missing bytes in request and/or response data. |
139 | | pub const MISSING_BYTES: u64 = 0x0008_0000_0000; |
140 | | /// Missing bytes in request data. |
141 | | pub const REQUEST_MISSING_BYTES: u64 = (0x0010_0000_0000 | Self::MISSING_BYTES); |
142 | | /// Missing bytes in the response data. |
143 | | pub const RESPONSE_MISSING_BYTES: u64 = (0x0020_0000_0000 | Self::MISSING_BYTES); |
144 | | /// Too many headers, log only once. |
145 | | pub const HEADERS_TOO_MANY: u64 = 0x0040_0000_0000; |
146 | | /// 100-Continue already seen. |
147 | | pub const FIELD_100_CONTINUE: u64 = 0x0080_0000_0000; |
148 | | /// Request chunk extension, log only once. |
149 | | pub const FIELD_REQ_CHUNK_EXTENSION: u64 = 0x0100_0000_0000; |
150 | | /// Response chunk extension, log only once. |
151 | | pub const FIELD_RESP_CHUNK_EXTENSION: u64 = 0x0200_0000_0000; |
152 | | } |
153 | | |
154 | | #[allow(clippy::upper_case_acronyms)] |
155 | | /// Enumerates possible EOLs |
156 | | #[derive(PartialEq, Eq, Copy, Clone, Debug)] |
157 | | pub(crate) enum Eol { |
158 | | /// '\n' |
159 | | LF, |
160 | | /// '\r' |
161 | | CR, |
162 | | /// "\r\n" |
163 | | CRLF, |
164 | | } |
165 | | |
166 | | /// Determines if character in a seperator. |
167 | | /// separators = "(" | ")" | "<" | ">" | "@" |
168 | | /// | "," | ";" | ":" | "\" | <"> |
169 | | /// | "/" | "[" | "]" | "?" | "=" |
170 | | /// | "{" | "}" | SP | HT |
171 | 11.7M | fn is_separator(c: u8) -> bool { |
172 | 10.3M | matches!( |
173 | 11.7M | c as char, |
174 | | '(' | ')' |
175 | | | '<' |
176 | | | '>' |
177 | | | '@' |
178 | | | ',' |
179 | | | ';' |
180 | | | ':' |
181 | | | '\\' |
182 | | | '"' |
183 | | | '/' |
184 | | | '[' |
185 | | | ']' |
186 | | | '?' |
187 | | | '=' |
188 | | | '{' |
189 | | | '}' |
190 | | | ' ' |
191 | | | '\t' |
192 | | ) |
193 | 11.7M | } |
194 | | |
195 | | /// Determines if character is a token. |
196 | | /// token = 1*<any CHAR except CTLs or separators> |
197 | | /// CHAR = <any US-ASCII character (octets 0 - 127)> |
198 | 16.6M | pub(crate) fn is_token(c: u8) -> bool { |
199 | 16.6M | (32..=126).contains(&c) && !is_separator(c) |
200 | 16.6M | } |
201 | | |
202 | | /// This parser takes leading whitespace as defined by is_ascii_whitespace. |
203 | 434k | pub(crate) fn take_ascii_whitespace() -> impl Fn(&[u8]) -> IResult<&[u8], &[u8]> { |
204 | 215k | move |input| take_while(|c: u8| c.is_ascii_whitespace())(input) |
205 | 434k | } |
206 | | |
207 | | /// Remove all line terminators (LF, CR or CRLF) from |
208 | | /// the end of the line provided as input. |
209 | 1.85M | pub(crate) fn chomp(mut data: &[u8]) -> &[u8] { |
210 | | loop { |
211 | 3.78M | let last_char = data.last(); |
212 | 3.78M | if last_char == Some(&(b'\n')) || last_char == Some(&(b'\r')) { |
213 | 1.92M | data = &data[..data.len() - 1]; |
214 | 1.92M | } else { |
215 | 1.85M | break; |
216 | | } |
217 | | } |
218 | 1.85M | data |
219 | 1.85M | } |
220 | | |
221 | | /// Trim the leading whitespace |
222 | 13.4M | fn trim_start(input: &[u8]) -> &[u8] { |
223 | 13.4M | let mut result = input; |
224 | 16.4M | while let Some(x) = result.first() { |
225 | 12.6M | if is_space(*x) { |
226 | 3.03M | result = &result[1..] |
227 | | } else { |
228 | 9.56M | break; |
229 | | } |
230 | | } |
231 | 13.4M | result |
232 | 13.4M | } |
233 | | |
234 | | /// Trim the trailing whitespace |
235 | 13.4M | fn trim_end(input: &[u8]) -> &[u8] { |
236 | 13.4M | let mut result = input; |
237 | 14.7M | while let Some(x) = result.last() { |
238 | 10.8M | if is_space(*x) { |
239 | 1.31M | result = &result[..(result.len() - 1)] |
240 | | } else { |
241 | 9.56M | break; |
242 | | } |
243 | | } |
244 | 13.4M | result |
245 | 13.4M | } |
246 | | |
247 | | /// Trim the leading and trailing whitespace from this byteslice. |
248 | 13.4M | pub(crate) fn trimmed(input: &[u8]) -> &[u8] { |
249 | 13.4M | trim_end(trim_start(input)) |
250 | 13.4M | } |
251 | | |
252 | | /// Splits the given input into two halves using the given predicate. |
253 | | /// The `reverse` parameter determines whether or not to split on the |
254 | | /// first match or the second match. |
255 | | /// The `do_trim` parameter will return results with leading and trailing |
256 | | /// whitespace trimmed. |
257 | | /// If the predicate does not match, then the entire input is returned |
258 | | /// in the first predicate element and an empty binary string is returned |
259 | | /// in the second element. |
260 | 273k | pub(crate) fn split_on_predicate<F>( |
261 | 273k | input: &[u8], reverse: bool, do_trim: bool, predicate: F, |
262 | 273k | ) -> (&[u8], &[u8]) |
263 | 273k | where |
264 | 273k | F: FnMut(&u8) -> bool, |
265 | | { |
266 | 273k | let (first, second) = if reverse { |
267 | 273k | let mut iter = input.rsplitn(2, predicate); |
268 | 273k | let mut second = iter.next(); |
269 | 273k | let mut first = iter.next(); |
270 | | // If we do not get two results, then put the only result first |
271 | 273k | if first.is_none() { |
272 | 158k | first = second; |
273 | 158k | second = None; |
274 | 158k | } |
275 | 273k | (first.unwrap_or(b""), second.unwrap_or(b"")) |
276 | | } else { |
277 | 0 | let mut iter = input.splitn(2, predicate); |
278 | 0 | let first = iter.next(); |
279 | 0 | let second = iter.next(); |
280 | 0 | (first.unwrap_or(b""), second.unwrap_or(b"")) |
281 | | }; |
282 | | |
283 | 273k | if do_trim { |
284 | 273k | (trimmed(first), trimmed(second)) |
285 | | } else { |
286 | 0 | (first, second) |
287 | | } |
288 | 273k | } suricata_htp::util::split_on_predicate::<<suricata_htp::connection_parser::ConnectionParser>::parse_request_line::{closure#3}>Line | Count | Source | 260 | 58.4k | pub(crate) fn split_on_predicate<F>( | 261 | 58.4k | input: &[u8], reverse: bool, do_trim: bool, predicate: F, | 262 | 58.4k | ) -> (&[u8], &[u8]) | 263 | 58.4k | where | 264 | 58.4k | F: FnMut(&u8) -> bool, | 265 | | { | 266 | 58.4k | let (first, second) = if reverse { | 267 | 58.4k | let mut iter = input.rsplitn(2, predicate); | 268 | 58.4k | let mut second = iter.next(); | 269 | 58.4k | let mut first = iter.next(); | 270 | | // If we do not get two results, then put the only result first | 271 | 58.4k | if first.is_none() { | 272 | 0 | first = second; | 273 | 0 | second = None; | 274 | 58.4k | } | 275 | 58.4k | (first.unwrap_or(b""), second.unwrap_or(b"")) | 276 | | } else { | 277 | 0 | let mut iter = input.splitn(2, predicate); | 278 | 0 | let first = iter.next(); | 279 | 0 | let second = iter.next(); | 280 | 0 | (first.unwrap_or(b""), second.unwrap_or(b"")) | 281 | | }; | 282 | | | 283 | 58.4k | if do_trim { | 284 | 58.4k | (trimmed(first), trimmed(second)) | 285 | | } else { | 286 | 0 | (first, second) | 287 | | } | 288 | 58.4k | } |
suricata_htp::util::split_on_predicate::<<suricata_htp::connection_parser::ConnectionParser>::parse_request_line::{closure#1}>Line | Count | Source | 260 | 215k | pub(crate) fn split_on_predicate<F>( | 261 | 215k | input: &[u8], reverse: bool, do_trim: bool, predicate: F, | 262 | 215k | ) -> (&[u8], &[u8]) | 263 | 215k | where | 264 | 215k | F: FnMut(&u8) -> bool, | 265 | | { | 266 | 215k | let (first, second) = if reverse { | 267 | 215k | let mut iter = input.rsplitn(2, predicate); | 268 | 215k | let mut second = iter.next(); | 269 | 215k | let mut first = iter.next(); | 270 | | // If we do not get two results, then put the only result first | 271 | 215k | if first.is_none() { | 272 | 158k | first = second; | 273 | 158k | second = None; | 274 | 158k | } | 275 | 215k | (first.unwrap_or(b""), second.unwrap_or(b"")) | 276 | | } else { | 277 | 0 | let mut iter = input.splitn(2, predicate); | 278 | 0 | let first = iter.next(); | 279 | 0 | let second = iter.next(); | 280 | 0 | (first.unwrap_or(b""), second.unwrap_or(b"")) | 281 | | }; | 282 | | | 283 | 215k | if do_trim { | 284 | 215k | (trimmed(first), trimmed(second)) | 285 | | } else { | 286 | 0 | (first, second) | 287 | | } | 288 | 215k | } |
|
289 | | |
290 | | /// Determines if character is a whitespace character. |
291 | | /// whitespace = ' ' | '\t' | '\r' | '\n' | '\x0b' | '\x0c' |
292 | 73.1M | pub(crate) fn is_space(c: u8) -> bool { |
293 | 73.1M | matches!(c as char, ' ' | '\t' | '\r' | '\n' | '\x0b' | '\x0c') |
294 | 73.1M | } |
295 | | |
296 | | /// Is the given line empty? |
297 | | /// |
298 | | /// Returns true or false |
299 | 4.10M | fn is_line_empty(data: &[u8]) -> bool { |
300 | 4.10M | matches!(data, b"\x0d" | b"\x0a" | b"\x0d\x0a") |
301 | 4.10M | } |
302 | | |
303 | | /// Determine if entire line is whitespace as defined by |
304 | | /// util::is_space. |
305 | 0 | fn is_line_whitespace(data: &[u8]) -> bool { |
306 | 0 | !data.iter().any(|c| !is_space(*c)) |
307 | 0 | } |
308 | | |
309 | | /// Searches for and extracts the next set of ascii digits from the input slice if present |
310 | | /// Parses over leading and trailing LWS characters. |
311 | | /// |
312 | | /// Returns (any trailing non-LWS characters, (non-LWS leading characters, ascii digits)) |
313 | 18.7k | pub(crate) fn ascii_digits(input: &[u8]) -> IResult<&[u8], (&[u8], &[u8])> { |
314 | 18.7k | map( |
315 | 18.7k | tuple(( |
316 | | nom_take_is_space, |
317 | 127k | take_till(|c: u8| c.is_ascii_digit()), |
318 | | digit1, |
319 | | nom_take_is_space, |
320 | | )), |
321 | 17.9k | |(_, leading_data, digits, _)| (leading_data, digits), |
322 | 18.7k | )(input) |
323 | 18.7k | } |
324 | | |
325 | | /// Searches for and extracts the next set of hex digits from the input slice if present |
326 | | /// Parses over leading and trailing LWS characters. |
327 | | /// |
328 | | /// Returns a tuple of any trailing non-LWS characters and the found hex digits |
329 | 0 | pub(crate) fn hex_digits() -> impl Fn(&[u8]) -> IResult<&[u8], &[u8]> { |
330 | 0 | move |input| { |
331 | 0 | map( |
332 | 0 | tuple(( |
333 | | nom_take_is_space, |
334 | 0 | take_while(|c: u8| c.is_ascii_hexdigit()), |
335 | | nom_take_is_space, |
336 | | )), |
337 | | |(_, digits, _)| digits, |
338 | 0 | )(input) |
339 | 0 | } |
340 | 0 | } |
341 | | |
342 | | /// Determines if the given line is a request terminator. |
343 | 4.10M | fn is_line_terminator( |
344 | 4.10M | server_personality: HtpServerPersonality, data: &[u8], next_no_lf: bool, |
345 | 4.10M | ) -> bool { |
346 | | // Is this the end of request headers? |
347 | 4.10M | if server_personality == HtpServerPersonality::IIS_5_0 { |
348 | | // IIS 5 will accept a whitespace line as a terminator |
349 | 0 | if is_line_whitespace(data) { |
350 | 0 | return true; |
351 | 0 | } |
352 | 4.10M | } |
353 | | |
354 | | // Treat an empty line as terminator |
355 | 4.10M | if is_line_empty(data) { |
356 | 1.69M | return true; |
357 | 2.40M | } |
358 | 2.40M | if data.len() == 2 && nom_is_space(data[0]) && data[1] == b'\n' { |
359 | 23.6k | return next_no_lf; |
360 | 2.38M | } |
361 | 2.38M | false |
362 | 4.10M | } |
363 | | |
364 | | /// Determines if the given line can be ignored when it appears before a request. |
365 | 4.10M | pub(crate) fn is_line_ignorable(server_personality: HtpServerPersonality, data: &[u8]) -> bool { |
366 | 4.10M | is_line_terminator(server_personality, data, false) |
367 | 4.10M | } |
368 | | |
369 | | /// Attempts to convert the provided port slice to a u16 |
370 | | /// |
371 | | /// Returns port number if a valid one is found. None if fails to convert or the result is 0 |
372 | 13.6k | pub(crate) fn convert_port(port: &[u8]) -> Option<u16> { |
373 | 13.6k | if port.is_empty() { |
374 | 0 | return None; |
375 | 13.6k | } |
376 | 13.6k | let port_number = std::str::from_utf8(port).ok()?.parse::<u16>().ok()?; |
377 | 1.88k | if port_number == 0 { |
378 | 14 | None |
379 | | } else { |
380 | 1.87k | Some(port_number) |
381 | | } |
382 | 13.6k | } |
383 | | |
384 | | /// Determine if the information provided on the response line |
385 | | /// is good enough. Browsers are lax when it comes to response |
386 | | /// line parsing. In most cases they will only look for the |
387 | | /// words "http" at the beginning. |
388 | | /// |
389 | | /// Returns true for good enough (treat as response body) or false for not good enough |
390 | 2.55M | pub(crate) fn treat_response_line_as_body(data: &[u8]) -> bool { |
391 | | // Browser behavior: |
392 | | // Firefox 3.5.x: (?i)^\s*http |
393 | | // IE: (?i)^\s*http\s*/ |
394 | | // Safari: ^HTTP/\d+\.\d+\s+\d{3} |
395 | | |
396 | 2.55M | tuple((opt(take_is_space_or_null), tag_no_case("http")))(data).is_err() |
397 | 2.55M | } |
398 | | |
399 | | /// Implements relaxed (not strictly RFC) hostname validation. |
400 | | /// |
401 | | /// Returns true if the supplied hostname is valid; false if it is not. |
402 | 78.5k | pub(crate) fn validate_hostname(input: &[u8]) -> bool { |
403 | 78.5k | if input.is_empty() || input.len() > 255 { |
404 | 1.65k | return false; |
405 | 76.9k | } |
406 | | |
407 | | // Check IPv6 |
408 | 76.9k | if let Ok((_rest, (_left_br, addr, _right_br))) = tuple(( |
409 | 76.9k | char::<_, NomError<&[u8]>>('['), |
410 | 76.9k | is_not::<_, _, NomError<&[u8]>>("#?/]"), |
411 | 76.9k | char::<_, NomError<&[u8]>>(']'), |
412 | 76.9k | ))(input) |
413 | | { |
414 | 468 | if let Ok(str) = std::str::from_utf8(addr) { |
415 | 253 | return std::net::Ipv6Addr::from_str(str).is_ok(); |
416 | 215 | } |
417 | 76.4k | } |
418 | | |
419 | 76.6k | if tag::<_, _, NomError<&[u8]>>(".")(input).is_ok() |
420 | 75.1k | || take_until::<_, _, NomError<&[u8]>>("..")(input).is_ok() |
421 | | { |
422 | 2.39k | return false; |
423 | 74.2k | } |
424 | 1.10M | for section in input.split(|&c| c == b'.') { |
425 | 83.1k | if section.len() > 63 { |
426 | 3.60k | return false; |
427 | 79.5k | } |
428 | | // According to the RFC, an underscore it not allowed in the label, but |
429 | | // we allow it here because we think it's often seen in practice. |
430 | 188k | if take_while_m_n::<_, _, NomError<&[u8]>>(section.len(), section.len(), |c| { |
431 | 188k | c == b'_' || c == b'-' || (c as char).is_alphanumeric() |
432 | 188k | })(section) |
433 | 79.5k | .is_err() |
434 | | { |
435 | 51.0k | return false; |
436 | 28.5k | } |
437 | | } |
438 | 19.6k | true |
439 | 78.5k | } |
440 | | |
441 | | /// Returns the LibHTP version string. |
442 | 0 | pub(crate) fn get_version() -> &'static str { |
443 | 0 | HTP_VERSION_STRING_FULL |
444 | 0 | } |
445 | | |
446 | | /// Take leading whitespace as defined by nom_is_space. |
447 | 36.7k | pub(crate) fn nom_take_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> { |
448 | 36.7k | take_while(nom_is_space)(data) |
449 | 36.7k | } |
450 | | |
451 | | /// Take data before the first null character if it exists. |
452 | 0 | pub(crate) fn take_until_null(data: &[u8]) -> IResult<&[u8], &[u8]> { |
453 | 0 | take_while(|c| c != b'\0')(data) |
454 | 0 | } |
455 | | |
456 | | /// Take leading space as defined by util::is_space. |
457 | 4.72M | pub(crate) fn take_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> { |
458 | 4.72M | take_while(is_space)(data) |
459 | 4.72M | } |
460 | | |
461 | | /// Take leading null characters or spaces as defined by util::is_space |
462 | 2.57M | pub(crate) fn take_is_space_or_null(data: &[u8]) -> IResult<&[u8], &[u8]> { |
463 | 3.51M | take_while(|c| is_space(c) || c == b'\0')(data) |
464 | 2.57M | } |
465 | | |
466 | | /// Take any non-space character as defined by is_space. |
467 | 3.98M | pub(crate) fn take_not_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> { |
468 | 27.7M | take_while(|c: u8| !is_space(c))(data) |
469 | 3.98M | } |
470 | | |
471 | | /// Returns all data up to and including the first new line or null |
472 | | /// Returns Err if not found |
473 | 1.53k | pub(crate) fn take_till_lf_null(data: &[u8]) -> IResult<&[u8], &[u8]> { |
474 | 189k | let (_, line) = streaming_take_till(|c| c == b'\n' || c == 0)(data)?; |
475 | 65 | Ok((&data[line.len() + 1..], &data[0..line.len() + 1])) |
476 | 1.53k | } |
477 | | |
478 | | /// Returns all data up to and including the first new line |
479 | | /// Returns Err if not found |
480 | 6.87M | pub(crate) fn take_till_lf(data: &[u8]) -> IResult<&[u8], &[u8]> { |
481 | 150M | let (_, line) = streaming_take_till(|c| c == b'\n')(data)?; |
482 | 6.50M | Ok((&data[line.len() + 1..], &data[0..line.len() + 1])) |
483 | 6.87M | } |
484 | | |
485 | | /// Returns all data up to and including the first EOL and which EOL was seen |
486 | | /// |
487 | | /// Returns Err if not found |
488 | 3.13M | pub(crate) fn take_till_eol(data: &[u8]) -> IResult<&[u8], (&[u8], Eol)> { |
489 | 3.13M | let (_, (line, eol)) = tuple(( |
490 | 36.9M | streaming_take_till(|c| c == b'\n' || c == b'\r'), |
491 | 3.13M | alt(( |
492 | 3.13M | streaming_tag("\r\n"), |
493 | 3.13M | streaming_tag("\r"), |
494 | 3.13M | streaming_tag("\n"), |
495 | 3.13M | )), |
496 | 3.13M | ))(data)?; |
497 | 2.83M | match eol { |
498 | 2.83M | b"\n" => Ok((&data[line.len() + 1..], (&data[0..line.len() + 1], Eol::LF))), |
499 | 587k | b"\r" => Ok((&data[line.len() + 1..], (&data[0..line.len() + 1], Eol::CR))), |
500 | 72.1k | b"\r\n" => Ok(( |
501 | 72.1k | &data[line.len() + 2..], |
502 | 72.1k | (&data[0..line.len() + 2], Eol::CRLF), |
503 | 72.1k | )), |
504 | 0 | _ => Err(Incomplete(Needed::new(1))), |
505 | | } |
506 | 3.13M | } |
507 | | |
508 | | /// Skip control characters |
509 | 0 | pub(crate) fn take_chunked_ctl_chars(data: &[u8]) -> IResult<&[u8], &[u8]> { |
510 | 0 | take_while(is_chunked_ctl_char)(data) |
511 | 0 | } |
512 | | |
513 | | /// Check if the data contains valid chunked length chars, i.e. leading chunked ctl chars and ascii hexdigits |
514 | | /// |
515 | | /// Returns true if valid, false otherwise |
516 | 0 | pub(crate) fn is_valid_chunked_length_data(data: &[u8]) -> bool { |
517 | 0 | tuple(( |
518 | | take_chunked_ctl_chars, |
519 | 0 | take_while1(|c: u8| !c.is_ascii_hexdigit()), |
520 | 0 | ))(data) |
521 | 0 | .is_err() |
522 | 0 | } |
523 | | |
524 | 0 | fn is_chunked_ctl_char(c: u8) -> bool { |
525 | 0 | matches!(c, 0x0d | 0x0a | 0x20 | 0x09 | 0x0b | 0x0c) |
526 | 0 | } |
527 | | |
528 | | /// Check if the entire input line is chunked control characters |
529 | 0 | pub(crate) fn is_chunked_ctl_line(l: &[u8]) -> bool { |
530 | 0 | for c in l { |
531 | 0 | if !is_chunked_ctl_char(*c) { |
532 | 0 | return false; |
533 | 0 | } |
534 | | } |
535 | 0 | true |
536 | 0 | } |
537 | | |
538 | | #[cfg(test)] |
539 | | mod tests { |
540 | | use crate::util::*; |
541 | | use rstest::rstest; |
542 | | |
543 | | #[rstest] |
544 | | #[case("", "", "")] |
545 | | #[case("hello world", "", "hello world")] |
546 | | #[case("\0", "\0", "")] |
547 | | #[case("hello_world \0 ", "\0 ", "hello_world ")] |
548 | | #[case("hello\0\0\0\0", "\0\0\0\0", "hello")] |
549 | | fn test_take_until_null(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) { |
550 | | assert_eq!( |
551 | | take_until_null(input.as_bytes()).unwrap(), |
552 | | (remaining.as_bytes(), parsed.as_bytes()) |
553 | | ); |
554 | | } |
555 | | |
556 | | #[rstest] |
557 | | #[case("", "", "")] |
558 | | #[case(" hell o", "hell o", " ")] |
559 | | #[case(" \thell o", "hell o", " \t")] |
560 | | #[case("hell o", "hell o", "")] |
561 | | #[case("\r\x0b \thell \to", "hell \to", "\r\x0b \t")] |
562 | | fn test_take_is_space(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) { |
563 | | assert_eq!( |
564 | | take_is_space(input.as_bytes()).unwrap(), |
565 | | (remaining.as_bytes(), parsed.as_bytes()) |
566 | | ); |
567 | | } |
568 | | |
569 | | #[rstest] |
570 | | #[case(" http 1.1", false)] |
571 | | #[case("\0 http 1.1", false)] |
572 | | #[case("http", false)] |
573 | | #[case("HTTP", false)] |
574 | | #[case(" HTTP", false)] |
575 | | #[case("test", true)] |
576 | | #[case(" test", true)] |
577 | | #[case("", true)] |
578 | | #[case("kfgjl hTtp ", true)] |
579 | | fn test_treat_response_line_as_body(#[case] input: &str, #[case] expected: bool) { |
580 | | assert_eq!(treat_response_line_as_body(input.as_bytes()), expected); |
581 | | } |
582 | | |
583 | | #[rstest] |
584 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
585 | | #[case("", "", "")] |
586 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
587 | | #[case("header:value\r\r", "", "")] |
588 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
589 | | #[case("header:value", "", "")] |
590 | | #[case("\nheader:value\r\n", "header:value\r\n", "\n")] |
591 | | #[case("header:value\r\n", "", "header:value\r\n")] |
592 | | #[case("header:value\n\r", "\r", "header:value\n")] |
593 | | #[case("header:value\n\n", "\n", "header:value\n")] |
594 | | #[case("abcdefg\nhijk", "hijk", "abcdefg\n")] |
595 | | fn test_take_till_lf(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) { |
596 | | assert_eq!( |
597 | | take_till_lf(input.as_bytes()).unwrap(), |
598 | | (remaining.as_bytes(), parsed.as_bytes()) |
599 | | ); |
600 | | } |
601 | | |
602 | | #[rstest] |
603 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
604 | | #[case("", "", "", Eol::CR)] |
605 | | #[case("abcdefg\n", "", "abcdefg\n", Eol::LF)] |
606 | | #[case("abcdefg\n\r", "\r", "abcdefg\n", Eol::LF)] |
607 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
608 | | #[case("abcdefg\r", "", "", Eol::CR)] |
609 | | #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")] |
610 | | #[case("abcdefg", "", "", Eol::CR)] |
611 | | #[case("abcdefg\nhijk", "hijk", "abcdefg\n", Eol::LF)] |
612 | | #[case("abcdefg\n\r\nhijk", "\r\nhijk", "abcdefg\n", Eol::LF)] |
613 | | #[case("abcdefg\rhijk", "hijk", "abcdefg\r", Eol::CR)] |
614 | | #[case("abcdefg\r\nhijk", "hijk", "abcdefg\r\n", Eol::CRLF)] |
615 | | #[case("abcdefg\r\n", "", "abcdefg\r\n", Eol::CRLF)] |
616 | | fn test_take_till_eol( |
617 | | #[case] input: &str, #[case] remaining: &str, #[case] parsed: &str, #[case] eol: Eol, |
618 | | ) { |
619 | | assert_eq!( |
620 | | take_till_eol(input.as_bytes()).unwrap(), |
621 | | (remaining.as_bytes(), (parsed.as_bytes(), eol)) |
622 | | ); |
623 | | } |
624 | | |
625 | | #[rstest] |
626 | | #[case(b'a', false)] |
627 | | #[case(b'^', false)] |
628 | | #[case(b'-', false)] |
629 | | #[case(b'_', false)] |
630 | | #[case(b'&', false)] |
631 | | #[case(b'(', true)] |
632 | | #[case(b'\\', true)] |
633 | | #[case(b'/', true)] |
634 | | #[case(b'=', true)] |
635 | | #[case(b'\t', true)] |
636 | | fn test_is_separator(#[case] input: u8, #[case] expected: bool) { |
637 | | assert_eq!(is_separator(input), expected); |
638 | | } |
639 | | |
640 | | #[rstest] |
641 | | #[case(b'a', true)] |
642 | | #[case(b'&', true)] |
643 | | #[case(b'+', true)] |
644 | | #[case(b'\t', false)] |
645 | | #[case(b'\n', false)] |
646 | | fn test_is_token(#[case] input: u8, #[case] expected: bool) { |
647 | | assert_eq!(is_token(input), expected); |
648 | | } |
649 | | |
650 | | #[rstest] |
651 | | #[case("", "")] |
652 | | #[case("test\n", "test")] |
653 | | #[case("test\r\n", "test")] |
654 | | #[case("test\r\n\n", "test")] |
655 | | #[case("test\n\r\r\n\r", "test")] |
656 | | #[case("test", "test")] |
657 | | #[case("te\nst", "te\nst")] |
658 | | fn test_chomp(#[case] input: &str, #[case] expected: &str) { |
659 | | assert_eq!(chomp(input.as_bytes()), expected.as_bytes()); |
660 | | } |
661 | | |
662 | | #[rstest] |
663 | | #[case::trimmed(b"notrim", b"notrim")] |
664 | | #[case::trim_start(b"\t trim", b"trim")] |
665 | | #[case::trim_both(b" trim ", b"trim")] |
666 | | #[case::trim_both_ignore_middle(b" trim trim ", b"trim trim")] |
667 | | #[case::trim_end(b"trim \t", b"trim")] |
668 | | #[case::trim_empty(b"", b"")] |
669 | | fn test_trim(#[case] input: &[u8], #[case] expected: &[u8]) { |
670 | | assert_eq!(trimmed(input), expected); |
671 | | } |
672 | | |
673 | | #[rstest] |
674 | | #[case::non_space(0x61, false)] |
675 | | #[case::space(0x20, true)] |
676 | | #[case::form_feed(0x0c, true)] |
677 | | #[case::newline(0x0a, true)] |
678 | | #[case::carriage_return(0x0d, true)] |
679 | | #[case::tab(0x09, true)] |
680 | | #[case::vertical_tab(0x0b, true)] |
681 | | fn test_is_space(#[case] input: u8, #[case] expected: bool) { |
682 | | assert_eq!(is_space(input), expected); |
683 | | } |
684 | | |
685 | | #[rstest] |
686 | | #[case("", false)] |
687 | | #[case("arfarf", false)] |
688 | | #[case("\n\r", false)] |
689 | | #[case("\rabc", false)] |
690 | | #[case("\r\n", true)] |
691 | | #[case("\r", true)] |
692 | | #[case("\n", true)] |
693 | | fn test_is_line_empty(#[case] input: &str, #[case] expected: bool) { |
694 | | assert_eq!(is_line_empty(input.as_bytes()), expected); |
695 | | } |
696 | | |
697 | | #[rstest] |
698 | | #[case("", false)] |
699 | | #[case("www.ExAmplE-1984.com", true)] |
700 | | #[case("[::]", true)] |
701 | | #[case("[2001:3db8:0000:0000:0000:ff00:d042:8530]", true)] |
702 | | #[case("www.example.com", true)] |
703 | | #[case("www.exa-mple.com", true)] |
704 | | #[case("www.exa_mple.com", true)] |
705 | | #[case(".www.example.com", false)] |
706 | | #[case("www..example.com", false)] |
707 | | #[case("www.example.com..", false)] |
708 | | #[case("www example com", false)] |
709 | | #[case("[::", false)] |
710 | | #[case("[::/path[0]", false)] |
711 | | #[case("[::#garbage]", false)] |
712 | | #[case("[::?]", false)] |
713 | | #[case::over64_char( |
714 | | "www.exampleexampleexampleexampleexampleexampleexampleexampleexampleexample.com", |
715 | | false |
716 | | )] |
717 | | fn test_validate_hostname(#[case] input: &str, #[case] expected: bool) { |
718 | | assert_eq!(validate_hostname(input.as_bytes()), expected); |
719 | | } |
720 | | |
721 | | #[rstest] |
722 | | #[should_panic( |
723 | | expected = "called `Result::unwrap()` on an `Err` value: Error(Error { input: [], code: Digit })" |
724 | | )] |
725 | | #[case(" garbage no ascii ", "", "", "")] |
726 | | #[case(" a200 \t bcd ", "bcd ", "a", "200")] |
727 | | #[case(" 555555555 ", "", "", "555555555")] |
728 | | #[case(" 555555555 500", "500", "", "555555555")] |
729 | | fn test_ascii_digits( |
730 | | #[case] input: &str, #[case] remaining: &str, #[case] leading: &str, #[case] digits: &str, |
731 | | ) { |
732 | | // Returns (any trailing non-LWS characters, (non-LWS leading characters, ascii digits)) |
733 | | assert_eq!( |
734 | | ascii_digits(input.as_bytes()).unwrap(), |
735 | | ( |
736 | | remaining.as_bytes(), |
737 | | (leading.as_bytes(), digits.as_bytes()) |
738 | | ) |
739 | | ); |
740 | | } |
741 | | |
742 | | #[rstest] |
743 | | #[case("", "", "")] |
744 | | #[case("12a5", "", "12a5")] |
745 | | #[case("12a5 .....", ".....", "12a5")] |
746 | | #[case(" \t12a5..... ", "..... ", "12a5")] |
747 | | #[case(" 68656c6c6f 12a5", "12a5", "68656c6c6f")] |
748 | | #[case(" .....", ".....", "")] |
749 | | fn test_hex_digits(#[case] input: &str, #[case] remaining: &str, #[case] digits: &str) { |
750 | | //(trailing non-LWS characters, found hex digits) |
751 | | assert_eq!( |
752 | | hex_digits()(input.as_bytes()).unwrap(), |
753 | | (remaining.as_bytes(), digits.as_bytes()) |
754 | | ); |
755 | | } |
756 | | |
757 | | #[rstest] |
758 | | #[case("", "", "")] |
759 | | #[case("no chunked ctl chars here", "no chunked ctl chars here", "")] |
760 | | #[case( |
761 | | "\x0d\x0a\x20\x09\x0b\x0cno chunked ctl chars here", |
762 | | "no chunked ctl chars here", |
763 | | "\x0d\x0a\x20\x09\x0b\x0c" |
764 | | )] |
765 | | #[case( |
766 | | "no chunked ctl chars here\x20\x09\x0b\x0c", |
767 | | "no chunked ctl chars here\x20\x09\x0b\x0c", |
768 | | "" |
769 | | )] |
770 | | #[case( |
771 | | "\x20\x09\x0b\x0cno chunked ctl chars here\x20\x09\x0b\x0c", |
772 | | "no chunked ctl chars here\x20\x09\x0b\x0c", |
773 | | "\x20\x09\x0b\x0c" |
774 | | )] |
775 | | fn test_take_chunked_ctl_chars( |
776 | | #[case] input: &str, #[case] remaining: &str, #[case] hex_digits: &str, |
777 | | ) { |
778 | | //(trailing non-LWS characters, found hex digits) |
779 | | assert_eq!( |
780 | | take_chunked_ctl_chars(input.as_bytes()).unwrap(), |
781 | | (remaining.as_bytes(), hex_digits.as_bytes()) |
782 | | ); |
783 | | } |
784 | | |
785 | | #[rstest] |
786 | | #[case("", true)] |
787 | | #[case("68656c6c6f", true)] |
788 | | #[case("\x0d\x0a\x20\x09\x0b\x0c68656c6c6f", true)] |
789 | | #[case("X5O!P%@AP", false)] |
790 | | #[case("\x0d\x0a\x20\x09\x0b\x0cX5O!P%@AP", false)] |
791 | | fn test_is_valid_chunked_length_data(#[case] input: &str, #[case] expected: bool) { |
792 | | assert_eq!(is_valid_chunked_length_data(input.as_bytes()), expected); |
793 | | } |
794 | | |
795 | | #[rstest] |
796 | | #[case("", false, true, ("", ""))] |
797 | | #[case("ONE TWO THREE", false, true, ("ONE", "TWO THREE"))] |
798 | | #[case("ONE TWO THREE", true, true, ("ONE TWO", "THREE"))] |
799 | | #[case("ONE TWO THREE", false, true, ("ONE", "TWO THREE"))] |
800 | | #[case("ONE TWO THREE", true, true, ("ONE TWO", "THREE"))] |
801 | | #[case("ONE", false, true, ("ONE", ""))] |
802 | | #[case("ONE", true, true, ("ONE", ""))] |
803 | | fn test_split_on_predicate( |
804 | | #[case] input: &str, #[case] reverse: bool, #[case] trim: bool, |
805 | | #[case] expected: (&str, &str), |
806 | | ) { |
807 | | assert_eq!( |
808 | | split_on_predicate(input.as_bytes(), reverse, trim, |c| *c == 0x20), |
809 | | (expected.0.as_bytes(), expected.1.as_bytes()) |
810 | | ); |
811 | | } |
812 | | } |