Coverage Report

Created: 2026-09-28 07:39

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/suricata8/rust/htp/src/util.rs
Line
Count
Source
1
//! Utility functions for http parsing.
2
3
use crate::{config::HtpServerPersonality, error::NomError};
4
use nom::{
5
    branch::alt,
6
    bytes::complete::{
7
        is_not, tag, tag_no_case, take_till, take_until, take_while, take_while1, take_while_m_n,
8
    },
9
    bytes::streaming::{tag as streaming_tag, take_till as streaming_take_till},
10
    character::complete::{char, digit1},
11
    character::is_space as nom_is_space,
12
    combinator::{map, opt},
13
    sequence::tuple,
14
    Err::Incomplete,
15
    IResult, Needed,
16
};
17
18
use std::str::FromStr;
19
20
/// String for the libhtp version.
21
pub const HTP_VERSION_STRING_FULL: &'_ str = concat!("LibHTP v", env!("CARGO_PKG_VERSION"), "\x00");
22
23
/// Trait to allow interacting with flags.
24
pub(crate) trait FlagOperations<T> {
25
    /// Inserts the specified flags in-place.
26
    fn set(&mut self, other: T);
27
    /// Removes the specified flags in-place.
28
    fn unset(&mut self, other: T);
29
    /// Determine if the specified flags are set
30
    fn is_set(&self, other: T) -> bool;
31
}
32
33
impl FlagOperations<u8> for u8 {
34
    /// Inserts the specified flags in-place.
35
589k
    fn set(&mut self, other: u8) {
36
589k
        *self |= other;
37
589k
    }
38
    /// Removes the specified flags in-place.
39
0
    fn unset(&mut self, other: u8) {
40
0
        *self &= !other;
41
0
    }
42
    /// Determine if the specified flags are set
43
0
    fn is_set(&self, other: u8) -> bool {
44
0
        self & other != 0
45
0
    }
46
}
47
48
impl FlagOperations<u64> for u64 {
49
    /// Inserts the specified flags in-place.
50
39.0M
    fn set(&mut self, other: u64) {
51
39.0M
        *self |= other;
52
39.0M
    }
53
    /// Removes the specified flags in-place.
54
31.9k
    fn unset(&mut self, other: u64) {
55
31.9k
        *self &= !other;
56
31.9k
    }
57
    /// Determine if the specified flags are set
58
24.2M
    fn is_set(&self, other: u64) -> bool {
59
24.2M
        self & other != 0
60
24.2M
    }
61
}
62
63
/// Various flag bits. Even though we have a flag field in several places
64
/// (header, transaction, connection), these fields are all in the same namespace
65
/// because we may want to set the same flag in several locations. For example, we
66
/// may set HTP_FIELD_FOLDED on the actual folded header, but also on the transaction
67
/// that contains the header. Both uses are useful.
68
#[repr(C)]
69
pub struct HtpFlags;
70
71
impl HtpFlags {
72
    /// Field cannot be parsed.
73
    pub const FIELD_UNPARSEABLE: u64 = 0x0000_0000_0004;
74
    /// Field is invalid.
75
    pub const FIELD_INVALID: u64 = 0x0000_0000_0008;
76
    /// Field is folded.
77
    pub const FIELD_FOLDED: u64 = 0x0000_0000_0010;
78
    /// Field has been seen more than once.
79
    pub const FIELD_REPEATED: u64 = 0x0000_0000_0020;
80
    // Field is too long.
81
    //pub const FIELD_LONG: u64 = 0x0000_0000_0040;
82
    // Field contains raw null byte.
83
    //pub const FIELD_RAW_NUL: u64 = 0x0000_0000_0080;
84
    /// Detect HTTP request smuggling.
85
    pub const REQUEST_SMUGGLING: u64 = 0x0000_0000_0100;
86
    /// Invalid header folding.
87
    pub const INVALID_FOLDING: u64 = 0x0000_0000_0200;
88
    /// Invalid request transfer-encoding.
89
    pub const REQUEST_INVALID_T_E: u64 = 0x0000_0000_0400;
90
    /// Multiple chunks.
91
    pub const MULTI_PACKET_HEAD: u64 = 0x0000_0000_0800;
92
    /// No host information in header.
93
    pub const HOST_MISSING: u64 = 0x0000_0000_1000;
94
    /// Inconsistent host or port information.
95
    pub const HOST_AMBIGUOUS: u64 = 0x0000_0000_2000;
96
    /// Encoded path contains null.
97
    pub const PATH_ENCODED_NUL: u64 = 0x0000_0000_4000;
98
    /// Url encoded contains raw null.
99
    pub const PATH_RAW_NUL: u64 = 0x0000_0000_8000;
100
    /// Url encoding is invalid.
101
    pub const PATH_INVALID_ENCODING: u64 = 0x0000_0001_0000;
102
    // Path is invalid.
103
    //pub const PATH_INVALID: u64 = 0x0000_0002_0000;
104
    /// Overlong usage in path.
105
    pub const PATH_OVERLONG_U: u64 = 0x0000_0004_0000;
106
    /// Encoded path separators present.
107
    pub const PATH_ENCODED_SEPARATOR: u64 = 0x0000_0008_0000;
108
    /// At least one valid UTF-8 character and no invalid ones.
109
    pub const PATH_UTF8_VALID: u64 = 0x0000_0010_0000;
110
    /// Invalid utf8 in path.
111
    pub const PATH_UTF8_INVALID: u64 = 0x0000_0020_0000;
112
    /// Invalid utf8 overlong character.
113
    pub const PATH_UTF8_OVERLONG: u64 = 0x0000_0040_0000;
114
    /// Range U+FF00 - U+FFEF detected.
115
    pub const PATH_HALF_FULL_RANGE: u64 = 0x0000_0080_0000;
116
    /// Status line is invalid.
117
    pub const STATUS_LINE_INVALID: u64 = 0x0000_0100_0000;
118
    /// Host in the URI.
119
    pub const HOSTU_INVALID: u64 = 0x0000_0200_0000;
120
    /// Host in the Host header.
121
    pub const HOSTH_INVALID: u64 = 0x0000_0400_0000;
122
    /// Contains null.
123
    pub const URLEN_ENCODED_NUL: u64 = 0x0000_0800_0000;
124
    /// Invalid encoding.
125
    pub const URLEN_INVALID_ENCODING: u64 = 0x0000_1000_0000;
126
    /// Overlong usage.
127
    pub const URLEN_OVERLONG_U: u64 = 0x0000_2000_0000;
128
    /// Range U+FF00 - U+FFEF detected.
129
    pub const URLEN_HALF_FULL_RANGE: u64 = 0x0000_4000_0000;
130
    /// Raw null byte.
131
    pub const URLEN_RAW_NUL: u64 = 0x0000_8000_0000;
132
    /// Request invalid.
133
    pub const REQUEST_INVALID: u64 = 0x0001_0000_0000;
134
    /// Request content-length invalid.
135
    pub const REQUEST_INVALID_C_L: u64 = 0x0002_0000_0000;
136
    /// Authorization is invalid.
137
    pub const AUTH_INVALID: u64 = 0x0004_0000_0000;
138
    /// Missing bytes in request and/or response data.
139
    pub const MISSING_BYTES: u64 = 0x0008_0000_0000;
140
    /// Missing bytes in request data.
141
    pub const REQUEST_MISSING_BYTES: u64 = (0x0010_0000_0000 | Self::MISSING_BYTES);
142
    /// Missing bytes in the response data.
143
    pub const RESPONSE_MISSING_BYTES: u64 = (0x0020_0000_0000 | Self::MISSING_BYTES);
144
    /// Too many headers, log only once.
145
    pub const HEADERS_TOO_MANY: u64 = 0x0040_0000_0000;
146
    /// 100-Continue already seen.
147
    pub const FIELD_100_CONTINUE: u64 = 0x0080_0000_0000;
148
    /// Request chunk extension, log only once.
149
    pub const FIELD_REQ_CHUNK_EXTENSION: u64 = 0x0100_0000_0000;
150
    /// Response chunk extension, log only once.
151
    pub const FIELD_RESP_CHUNK_EXTENSION: u64 = 0x0200_0000_0000;
152
}
153
154
#[allow(clippy::upper_case_acronyms)]
155
/// Enumerates possible EOLs
156
#[derive(PartialEq, Eq, Copy, Clone, Debug)]
157
pub(crate) enum Eol {
158
    /// '\n'
159
    LF,
160
    /// '\r'
161
    CR,
162
    /// "\r\n"
163
    CRLF,
164
}
165
166
/// Determines if character in a seperator.
167
/// separators = "(" | ")" | "<" | ">" | "@"
168
/// | "," | ";" | ":" | "\" | <">
169
/// | "/" | "[" | "]" | "?" | "="
170
/// | "{" | "}" | SP | HT
171
11.7M
fn is_separator(c: u8) -> bool {
172
10.3M
    matches!(
173
11.7M
        c as char,
174
        '(' | ')'
175
            | '<'
176
            | '>'
177
            | '@'
178
            | ','
179
            | ';'
180
            | ':'
181
            | '\\'
182
            | '"'
183
            | '/'
184
            | '['
185
            | ']'
186
            | '?'
187
            | '='
188
            | '{'
189
            | '}'
190
            | ' '
191
            | '\t'
192
    )
193
11.7M
}
194
195
/// Determines if character is a token.
196
/// token = 1*<any CHAR except CTLs or separators>
197
/// CHAR  = <any US-ASCII character (octets 0 - 127)>
198
16.6M
pub(crate) fn is_token(c: u8) -> bool {
199
16.6M
    (32..=126).contains(&c) && !is_separator(c)
200
16.6M
}
201
202
/// This parser takes leading whitespace as defined by is_ascii_whitespace.
203
434k
pub(crate) fn take_ascii_whitespace() -> impl Fn(&[u8]) -> IResult<&[u8], &[u8]> {
204
215k
    move |input| take_while(|c: u8| c.is_ascii_whitespace())(input)
205
434k
}
206
207
/// Remove all line terminators (LF, CR or CRLF) from
208
/// the end of the line provided as input.
209
1.85M
pub(crate) fn chomp(mut data: &[u8]) -> &[u8] {
210
    loop {
211
3.78M
        let last_char = data.last();
212
3.78M
        if last_char == Some(&(b'\n')) || last_char == Some(&(b'\r')) {
213
1.92M
            data = &data[..data.len() - 1];
214
1.92M
        } else {
215
1.85M
            break;
216
        }
217
    }
218
1.85M
    data
219
1.85M
}
220
221
/// Trim the leading whitespace
222
13.4M
fn trim_start(input: &[u8]) -> &[u8] {
223
13.4M
    let mut result = input;
224
16.4M
    while let Some(x) = result.first() {
225
12.6M
        if is_space(*x) {
226
3.03M
            result = &result[1..]
227
        } else {
228
9.56M
            break;
229
        }
230
    }
231
13.4M
    result
232
13.4M
}
233
234
/// Trim the trailing whitespace
235
13.4M
fn trim_end(input: &[u8]) -> &[u8] {
236
13.4M
    let mut result = input;
237
14.7M
    while let Some(x) = result.last() {
238
10.8M
        if is_space(*x) {
239
1.31M
            result = &result[..(result.len() - 1)]
240
        } else {
241
9.56M
            break;
242
        }
243
    }
244
13.4M
    result
245
13.4M
}
246
247
/// Trim the leading and trailing whitespace from this byteslice.
248
13.4M
pub(crate) fn trimmed(input: &[u8]) -> &[u8] {
249
13.4M
    trim_end(trim_start(input))
250
13.4M
}
251
252
/// Splits the given input into two halves using the given predicate.
253
/// The `reverse` parameter determines whether or not to split on the
254
/// first match or the second match.
255
/// The `do_trim` parameter will return results with leading and trailing
256
/// whitespace trimmed.
257
/// If the predicate does not match, then the entire input is returned
258
/// in the first predicate element and an empty binary string is returned
259
/// in the second element.
260
273k
pub(crate) fn split_on_predicate<F>(
261
273k
    input: &[u8], reverse: bool, do_trim: bool, predicate: F,
262
273k
) -> (&[u8], &[u8])
263
273k
where
264
273k
    F: FnMut(&u8) -> bool,
265
{
266
273k
    let (first, second) = if reverse {
267
273k
        let mut iter = input.rsplitn(2, predicate);
268
273k
        let mut second = iter.next();
269
273k
        let mut first = iter.next();
270
        // If we do not get two results, then put the only result first
271
273k
        if first.is_none() {
272
158k
            first = second;
273
158k
            second = None;
274
158k
        }
275
273k
        (first.unwrap_or(b""), second.unwrap_or(b""))
276
    } else {
277
0
        let mut iter = input.splitn(2, predicate);
278
0
        let first = iter.next();
279
0
        let second = iter.next();
280
0
        (first.unwrap_or(b""), second.unwrap_or(b""))
281
    };
282
283
273k
    if do_trim {
284
273k
        (trimmed(first), trimmed(second))
285
    } else {
286
0
        (first, second)
287
    }
288
273k
}
suricata_htp::util::split_on_predicate::<<suricata_htp::connection_parser::ConnectionParser>::parse_request_line::{closure#3}>
Line
Count
Source
260
58.4k
pub(crate) fn split_on_predicate<F>(
261
58.4k
    input: &[u8], reverse: bool, do_trim: bool, predicate: F,
262
58.4k
) -> (&[u8], &[u8])
263
58.4k
where
264
58.4k
    F: FnMut(&u8) -> bool,
265
{
266
58.4k
    let (first, second) = if reverse {
267
58.4k
        let mut iter = input.rsplitn(2, predicate);
268
58.4k
        let mut second = iter.next();
269
58.4k
        let mut first = iter.next();
270
        // If we do not get two results, then put the only result first
271
58.4k
        if first.is_none() {
272
0
            first = second;
273
0
            second = None;
274
58.4k
        }
275
58.4k
        (first.unwrap_or(b""), second.unwrap_or(b""))
276
    } else {
277
0
        let mut iter = input.splitn(2, predicate);
278
0
        let first = iter.next();
279
0
        let second = iter.next();
280
0
        (first.unwrap_or(b""), second.unwrap_or(b""))
281
    };
282
283
58.4k
    if do_trim {
284
58.4k
        (trimmed(first), trimmed(second))
285
    } else {
286
0
        (first, second)
287
    }
288
58.4k
}
suricata_htp::util::split_on_predicate::<<suricata_htp::connection_parser::ConnectionParser>::parse_request_line::{closure#1}>
Line
Count
Source
260
215k
pub(crate) fn split_on_predicate<F>(
261
215k
    input: &[u8], reverse: bool, do_trim: bool, predicate: F,
262
215k
) -> (&[u8], &[u8])
263
215k
where
264
215k
    F: FnMut(&u8) -> bool,
265
{
266
215k
    let (first, second) = if reverse {
267
215k
        let mut iter = input.rsplitn(2, predicate);
268
215k
        let mut second = iter.next();
269
215k
        let mut first = iter.next();
270
        // If we do not get two results, then put the only result first
271
215k
        if first.is_none() {
272
158k
            first = second;
273
158k
            second = None;
274
158k
        }
275
215k
        (first.unwrap_or(b""), second.unwrap_or(b""))
276
    } else {
277
0
        let mut iter = input.splitn(2, predicate);
278
0
        let first = iter.next();
279
0
        let second = iter.next();
280
0
        (first.unwrap_or(b""), second.unwrap_or(b""))
281
    };
282
283
215k
    if do_trim {
284
215k
        (trimmed(first), trimmed(second))
285
    } else {
286
0
        (first, second)
287
    }
288
215k
}
289
290
/// Determines if character is a whitespace character.
291
/// whitespace = ' ' | '\t' | '\r' | '\n' | '\x0b' | '\x0c'
292
73.1M
pub(crate) fn is_space(c: u8) -> bool {
293
73.1M
    matches!(c as char, ' ' | '\t' | '\r' | '\n' | '\x0b' | '\x0c')
294
73.1M
}
295
296
/// Is the given line empty?
297
///
298
/// Returns true or false
299
4.10M
fn is_line_empty(data: &[u8]) -> bool {
300
4.10M
    matches!(data, b"\x0d" | b"\x0a" | b"\x0d\x0a")
301
4.10M
}
302
303
/// Determine if entire line is whitespace as defined by
304
/// util::is_space.
305
0
fn is_line_whitespace(data: &[u8]) -> bool {
306
0
    !data.iter().any(|c| !is_space(*c))
307
0
}
308
309
/// Searches for and extracts the next set of ascii digits from the input slice if present
310
/// Parses over leading and trailing LWS characters.
311
///
312
/// Returns (any trailing non-LWS characters, (non-LWS leading characters, ascii digits))
313
18.7k
pub(crate) fn ascii_digits(input: &[u8]) -> IResult<&[u8], (&[u8], &[u8])> {
314
18.7k
    map(
315
18.7k
        tuple((
316
            nom_take_is_space,
317
127k
            take_till(|c: u8| c.is_ascii_digit()),
318
            digit1,
319
            nom_take_is_space,
320
        )),
321
17.9k
        |(_, leading_data, digits, _)| (leading_data, digits),
322
18.7k
    )(input)
323
18.7k
}
324
325
/// Searches for and extracts the next set of hex digits from the input slice if present
326
/// Parses over leading and trailing LWS characters.
327
///
328
/// Returns a tuple of any trailing non-LWS characters and the found hex digits
329
0
pub(crate) fn hex_digits() -> impl Fn(&[u8]) -> IResult<&[u8], &[u8]> {
330
0
    move |input| {
331
0
        map(
332
0
            tuple((
333
                nom_take_is_space,
334
0
                take_while(|c: u8| c.is_ascii_hexdigit()),
335
                nom_take_is_space,
336
            )),
337
            |(_, digits, _)| digits,
338
0
        )(input)
339
0
    }
340
0
}
341
342
/// Determines if the given line is a request terminator.
343
4.10M
fn is_line_terminator(
344
4.10M
    server_personality: HtpServerPersonality, data: &[u8], next_no_lf: bool,
345
4.10M
) -> bool {
346
    // Is this the end of request headers?
347
4.10M
    if server_personality == HtpServerPersonality::IIS_5_0 {
348
        // IIS 5 will accept a whitespace line as a terminator
349
0
        if is_line_whitespace(data) {
350
0
            return true;
351
0
        }
352
4.10M
    }
353
354
    // Treat an empty line as terminator
355
4.10M
    if is_line_empty(data) {
356
1.69M
        return true;
357
2.40M
    }
358
2.40M
    if data.len() == 2 && nom_is_space(data[0]) && data[1] == b'\n' {
359
23.6k
        return next_no_lf;
360
2.38M
    }
361
2.38M
    false
362
4.10M
}
363
364
/// Determines if the given line can be ignored when it appears before a request.
365
4.10M
pub(crate) fn is_line_ignorable(server_personality: HtpServerPersonality, data: &[u8]) -> bool {
366
4.10M
    is_line_terminator(server_personality, data, false)
367
4.10M
}
368
369
/// Attempts to convert the provided port slice to a u16
370
///
371
/// Returns port number if a valid one is found. None if fails to convert or the result is 0
372
13.6k
pub(crate) fn convert_port(port: &[u8]) -> Option<u16> {
373
13.6k
    if port.is_empty() {
374
0
        return None;
375
13.6k
    }
376
13.6k
    let port_number = std::str::from_utf8(port).ok()?.parse::<u16>().ok()?;
377
1.88k
    if port_number == 0 {
378
14
        None
379
    } else {
380
1.87k
        Some(port_number)
381
    }
382
13.6k
}
383
384
/// Determine if the information provided on the response line
385
/// is good enough. Browsers are lax when it comes to response
386
/// line parsing. In most cases they will only look for the
387
/// words "http" at the beginning.
388
///
389
/// Returns true for good enough (treat as response body) or false for not good enough
390
2.55M
pub(crate) fn treat_response_line_as_body(data: &[u8]) -> bool {
391
    // Browser behavior:
392
    //      Firefox 3.5.x: (?i)^\s*http
393
    //      IE: (?i)^\s*http\s*/
394
    //      Safari: ^HTTP/\d+\.\d+\s+\d{3}
395
396
2.55M
    tuple((opt(take_is_space_or_null), tag_no_case("http")))(data).is_err()
397
2.55M
}
398
399
/// Implements relaxed (not strictly RFC) hostname validation.
400
///
401
/// Returns true if the supplied hostname is valid; false if it is not.
402
78.5k
pub(crate) fn validate_hostname(input: &[u8]) -> bool {
403
78.5k
    if input.is_empty() || input.len() > 255 {
404
1.65k
        return false;
405
76.9k
    }
406
407
    // Check IPv6
408
76.9k
    if let Ok((_rest, (_left_br, addr, _right_br))) = tuple((
409
76.9k
        char::<_, NomError<&[u8]>>('['),
410
76.9k
        is_not::<_, _, NomError<&[u8]>>("#?/]"),
411
76.9k
        char::<_, NomError<&[u8]>>(']'),
412
76.9k
    ))(input)
413
    {
414
468
        if let Ok(str) = std::str::from_utf8(addr) {
415
253
            return std::net::Ipv6Addr::from_str(str).is_ok();
416
215
        }
417
76.4k
    }
418
419
76.6k
    if tag::<_, _, NomError<&[u8]>>(".")(input).is_ok()
420
75.1k
        || take_until::<_, _, NomError<&[u8]>>("..")(input).is_ok()
421
    {
422
2.39k
        return false;
423
74.2k
    }
424
1.10M
    for section in input.split(|&c| c == b'.') {
425
83.1k
        if section.len() > 63 {
426
3.60k
            return false;
427
79.5k
        }
428
        // According to the RFC, an underscore it not allowed in the label, but
429
        // we allow it here because we think it's often seen in practice.
430
188k
        if take_while_m_n::<_, _, NomError<&[u8]>>(section.len(), section.len(), |c| {
431
188k
            c == b'_' || c == b'-' || (c as char).is_alphanumeric()
432
188k
        })(section)
433
79.5k
        .is_err()
434
        {
435
51.0k
            return false;
436
28.5k
        }
437
    }
438
19.6k
    true
439
78.5k
}
440
441
/// Returns the LibHTP version string.
442
0
pub(crate) fn get_version() -> &'static str {
443
0
    HTP_VERSION_STRING_FULL
444
0
}
445
446
/// Take leading whitespace as defined by nom_is_space.
447
36.7k
pub(crate) fn nom_take_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> {
448
36.7k
    take_while(nom_is_space)(data)
449
36.7k
}
450
451
/// Take data before the first null character if it exists.
452
0
pub(crate) fn take_until_null(data: &[u8]) -> IResult<&[u8], &[u8]> {
453
0
    take_while(|c| c != b'\0')(data)
454
0
}
455
456
/// Take leading space as defined by util::is_space.
457
4.72M
pub(crate) fn take_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> {
458
4.72M
    take_while(is_space)(data)
459
4.72M
}
460
461
/// Take leading null characters or spaces as defined by util::is_space
462
2.57M
pub(crate) fn take_is_space_or_null(data: &[u8]) -> IResult<&[u8], &[u8]> {
463
3.51M
    take_while(|c| is_space(c) || c == b'\0')(data)
464
2.57M
}
465
466
/// Take any non-space character as defined by is_space.
467
3.98M
pub(crate) fn take_not_is_space(data: &[u8]) -> IResult<&[u8], &[u8]> {
468
27.7M
    take_while(|c: u8| !is_space(c))(data)
469
3.98M
}
470
471
/// Returns all data up to and including the first new line or null
472
/// Returns Err if not found
473
1.53k
pub(crate) fn take_till_lf_null(data: &[u8]) -> IResult<&[u8], &[u8]> {
474
189k
    let (_, line) = streaming_take_till(|c| c == b'\n' || c == 0)(data)?;
475
65
    Ok((&data[line.len() + 1..], &data[0..line.len() + 1]))
476
1.53k
}
477
478
/// Returns all data up to and including the first new line
479
/// Returns Err if not found
480
6.87M
pub(crate) fn take_till_lf(data: &[u8]) -> IResult<&[u8], &[u8]> {
481
150M
    let (_, line) = streaming_take_till(|c| c == b'\n')(data)?;
482
6.50M
    Ok((&data[line.len() + 1..], &data[0..line.len() + 1]))
483
6.87M
}
484
485
/// Returns all data up to and including the first EOL and which EOL was seen
486
///
487
/// Returns Err if not found
488
3.13M
pub(crate) fn take_till_eol(data: &[u8]) -> IResult<&[u8], (&[u8], Eol)> {
489
3.13M
    let (_, (line, eol)) = tuple((
490
36.9M
        streaming_take_till(|c| c == b'\n' || c == b'\r'),
491
3.13M
        alt((
492
3.13M
            streaming_tag("\r\n"),
493
3.13M
            streaming_tag("\r"),
494
3.13M
            streaming_tag("\n"),
495
3.13M
        )),
496
3.13M
    ))(data)?;
497
2.83M
    match eol {
498
2.83M
        b"\n" => Ok((&data[line.len() + 1..], (&data[0..line.len() + 1], Eol::LF))),
499
587k
        b"\r" => Ok((&data[line.len() + 1..], (&data[0..line.len() + 1], Eol::CR))),
500
72.1k
        b"\r\n" => Ok((
501
72.1k
            &data[line.len() + 2..],
502
72.1k
            (&data[0..line.len() + 2], Eol::CRLF),
503
72.1k
        )),
504
0
        _ => Err(Incomplete(Needed::new(1))),
505
    }
506
3.13M
}
507
508
/// Skip control characters
509
0
pub(crate) fn take_chunked_ctl_chars(data: &[u8]) -> IResult<&[u8], &[u8]> {
510
0
    take_while(is_chunked_ctl_char)(data)
511
0
}
512
513
/// Check if the data contains valid chunked length chars, i.e. leading chunked ctl chars and ascii hexdigits
514
///
515
/// Returns true if valid, false otherwise
516
0
pub(crate) fn is_valid_chunked_length_data(data: &[u8]) -> bool {
517
0
    tuple((
518
        take_chunked_ctl_chars,
519
0
        take_while1(|c: u8| !c.is_ascii_hexdigit()),
520
0
    ))(data)
521
0
    .is_err()
522
0
}
523
524
0
fn is_chunked_ctl_char(c: u8) -> bool {
525
0
    matches!(c, 0x0d | 0x0a | 0x20 | 0x09 | 0x0b | 0x0c)
526
0
}
527
528
/// Check if the entire input line is chunked control characters
529
0
pub(crate) fn is_chunked_ctl_line(l: &[u8]) -> bool {
530
0
    for c in l {
531
0
        if !is_chunked_ctl_char(*c) {
532
0
            return false;
533
0
        }
534
    }
535
0
    true
536
0
}
537
538
#[cfg(test)]
539
mod tests {
540
    use crate::util::*;
541
    use rstest::rstest;
542
543
    #[rstest]
544
    #[case("", "", "")]
545
    #[case("hello world", "", "hello world")]
546
    #[case("\0", "\0", "")]
547
    #[case("hello_world  \0   ", "\0   ", "hello_world  ")]
548
    #[case("hello\0\0\0\0", "\0\0\0\0", "hello")]
549
    fn test_take_until_null(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) {
550
        assert_eq!(
551
            take_until_null(input.as_bytes()).unwrap(),
552
            (remaining.as_bytes(), parsed.as_bytes())
553
        );
554
    }
555
556
    #[rstest]
557
    #[case("", "", "")]
558
    #[case("   hell o", "hell o", "   ")]
559
    #[case("   \thell o", "hell o", "   \t")]
560
    #[case("hell o", "hell o", "")]
561
    #[case("\r\x0b  \thell \to", "hell \to", "\r\x0b  \t")]
562
    fn test_take_is_space(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) {
563
        assert_eq!(
564
            take_is_space(input.as_bytes()).unwrap(),
565
            (remaining.as_bytes(), parsed.as_bytes())
566
        );
567
    }
568
569
    #[rstest]
570
    #[case("   http 1.1", false)]
571
    #[case("\0 http 1.1", false)]
572
    #[case("http", false)]
573
    #[case("HTTP", false)]
574
    #[case("    HTTP", false)]
575
    #[case("test", true)]
576
    #[case("     test", true)]
577
    #[case("", true)]
578
    #[case("kfgjl  hTtp ", true)]
579
    fn test_treat_response_line_as_body(#[case] input: &str, #[case] expected: bool) {
580
        assert_eq!(treat_response_line_as_body(input.as_bytes()), expected);
581
    }
582
583
    #[rstest]
584
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
585
    #[case("", "", "")]
586
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
587
    #[case("header:value\r\r", "", "")]
588
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
589
    #[case("header:value", "", "")]
590
    #[case("\nheader:value\r\n", "header:value\r\n", "\n")]
591
    #[case("header:value\r\n", "", "header:value\r\n")]
592
    #[case("header:value\n\r", "\r", "header:value\n")]
593
    #[case("header:value\n\n", "\n", "header:value\n")]
594
    #[case("abcdefg\nhijk", "hijk", "abcdefg\n")]
595
    fn test_take_till_lf(#[case] input: &str, #[case] remaining: &str, #[case] parsed: &str) {
596
        assert_eq!(
597
            take_till_lf(input.as_bytes()).unwrap(),
598
            (remaining.as_bytes(), parsed.as_bytes())
599
        );
600
    }
601
602
    #[rstest]
603
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
604
    #[case("", "", "", Eol::CR)]
605
    #[case("abcdefg\n", "", "abcdefg\n", Eol::LF)]
606
    #[case("abcdefg\n\r", "\r", "abcdefg\n", Eol::LF)]
607
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
608
    #[case("abcdefg\r", "", "", Eol::CR)]
609
    #[should_panic(expected = "called `Result::unwrap()` on an `Err` value: Incomplete(Size(1))")]
610
    #[case("abcdefg", "", "", Eol::CR)]
611
    #[case("abcdefg\nhijk", "hijk", "abcdefg\n", Eol::LF)]
612
    #[case("abcdefg\n\r\nhijk", "\r\nhijk", "abcdefg\n", Eol::LF)]
613
    #[case("abcdefg\rhijk", "hijk", "abcdefg\r", Eol::CR)]
614
    #[case("abcdefg\r\nhijk", "hijk", "abcdefg\r\n", Eol::CRLF)]
615
    #[case("abcdefg\r\n", "", "abcdefg\r\n", Eol::CRLF)]
616
    fn test_take_till_eol(
617
        #[case] input: &str, #[case] remaining: &str, #[case] parsed: &str, #[case] eol: Eol,
618
    ) {
619
        assert_eq!(
620
            take_till_eol(input.as_bytes()).unwrap(),
621
            (remaining.as_bytes(), (parsed.as_bytes(), eol))
622
        );
623
    }
624
625
    #[rstest]
626
    #[case(b'a', false)]
627
    #[case(b'^', false)]
628
    #[case(b'-', false)]
629
    #[case(b'_', false)]
630
    #[case(b'&', false)]
631
    #[case(b'(', true)]
632
    #[case(b'\\', true)]
633
    #[case(b'/', true)]
634
    #[case(b'=', true)]
635
    #[case(b'\t', true)]
636
    fn test_is_separator(#[case] input: u8, #[case] expected: bool) {
637
        assert_eq!(is_separator(input), expected);
638
    }
639
640
    #[rstest]
641
    #[case(b'a', true)]
642
    #[case(b'&', true)]
643
    #[case(b'+', true)]
644
    #[case(b'\t', false)]
645
    #[case(b'\n', false)]
646
    fn test_is_token(#[case] input: u8, #[case] expected: bool) {
647
        assert_eq!(is_token(input), expected);
648
    }
649
650
    #[rstest]
651
    #[case("", "")]
652
    #[case("test\n", "test")]
653
    #[case("test\r\n", "test")]
654
    #[case("test\r\n\n", "test")]
655
    #[case("test\n\r\r\n\r", "test")]
656
    #[case("test", "test")]
657
    #[case("te\nst", "te\nst")]
658
    fn test_chomp(#[case] input: &str, #[case] expected: &str) {
659
        assert_eq!(chomp(input.as_bytes()), expected.as_bytes());
660
    }
661
662
    #[rstest]
663
    #[case::trimmed(b"notrim", b"notrim")]
664
    #[case::trim_start(b"\t trim", b"trim")]
665
    #[case::trim_both(b" trim ", b"trim")]
666
    #[case::trim_both_ignore_middle(b" trim trim ", b"trim trim")]
667
    #[case::trim_end(b"trim \t", b"trim")]
668
    #[case::trim_empty(b"", b"")]
669
    fn test_trim(#[case] input: &[u8], #[case] expected: &[u8]) {
670
        assert_eq!(trimmed(input), expected);
671
    }
672
673
    #[rstest]
674
    #[case::non_space(0x61, false)]
675
    #[case::space(0x20, true)]
676
    #[case::form_feed(0x0c, true)]
677
    #[case::newline(0x0a, true)]
678
    #[case::carriage_return(0x0d, true)]
679
    #[case::tab(0x09, true)]
680
    #[case::vertical_tab(0x0b, true)]
681
    fn test_is_space(#[case] input: u8, #[case] expected: bool) {
682
        assert_eq!(is_space(input), expected);
683
    }
684
685
    #[rstest]
686
    #[case("", false)]
687
    #[case("arfarf", false)]
688
    #[case("\n\r", false)]
689
    #[case("\rabc", false)]
690
    #[case("\r\n", true)]
691
    #[case("\r", true)]
692
    #[case("\n", true)]
693
    fn test_is_line_empty(#[case] input: &str, #[case] expected: bool) {
694
        assert_eq!(is_line_empty(input.as_bytes()), expected);
695
    }
696
697
    #[rstest]
698
    #[case("", false)]
699
    #[case("www.ExAmplE-1984.com", true)]
700
    #[case("[::]", true)]
701
    #[case("[2001:3db8:0000:0000:0000:ff00:d042:8530]", true)]
702
    #[case("www.example.com", true)]
703
    #[case("www.exa-mple.com", true)]
704
    #[case("www.exa_mple.com", true)]
705
    #[case(".www.example.com", false)]
706
    #[case("www..example.com", false)]
707
    #[case("www.example.com..", false)]
708
    #[case("www example com", false)]
709
    #[case("[::", false)]
710
    #[case("[::/path[0]", false)]
711
    #[case("[::#garbage]", false)]
712
    #[case("[::?]", false)]
713
    #[case::over64_char(
714
        "www.exampleexampleexampleexampleexampleexampleexampleexampleexampleexample.com",
715
        false
716
    )]
717
    fn test_validate_hostname(#[case] input: &str, #[case] expected: bool) {
718
        assert_eq!(validate_hostname(input.as_bytes()), expected);
719
    }
720
721
    #[rstest]
722
    #[should_panic(
723
        expected = "called `Result::unwrap()` on an `Err` value: Error(Error { input: [], code: Digit })"
724
    )]
725
    #[case("   garbage no ascii ", "", "", "")]
726
    #[case("    a200 \t  bcd ", "bcd ", "a", "200")]
727
    #[case("   555555555    ", "", "", "555555555")]
728
    #[case("   555555555    500", "500", "", "555555555")]
729
    fn test_ascii_digits(
730
        #[case] input: &str, #[case] remaining: &str, #[case] leading: &str, #[case] digits: &str,
731
    ) {
732
        // Returns (any trailing non-LWS characters, (non-LWS leading characters, ascii digits))
733
        assert_eq!(
734
            ascii_digits(input.as_bytes()).unwrap(),
735
            (
736
                remaining.as_bytes(),
737
                (leading.as_bytes(), digits.as_bytes())
738
            )
739
        );
740
    }
741
742
    #[rstest]
743
    #[case("", "", "")]
744
    #[case("12a5", "", "12a5")]
745
    #[case("12a5   .....", ".....", "12a5")]
746
    #[case("    \t12a5.....    ", ".....    ", "12a5")]
747
    #[case(" 68656c6c6f   12a5", "12a5", "68656c6c6f")]
748
    #[case("  .....", ".....", "")]
749
    fn test_hex_digits(#[case] input: &str, #[case] remaining: &str, #[case] digits: &str) {
750
        //(trailing non-LWS characters, found hex digits)
751
        assert_eq!(
752
            hex_digits()(input.as_bytes()).unwrap(),
753
            (remaining.as_bytes(), digits.as_bytes())
754
        );
755
    }
756
757
    #[rstest]
758
    #[case("", "", "")]
759
    #[case("no chunked ctl chars here", "no chunked ctl chars here", "")]
760
    #[case(
761
        "\x0d\x0a\x20\x09\x0b\x0cno chunked ctl chars here",
762
        "no chunked ctl chars here",
763
        "\x0d\x0a\x20\x09\x0b\x0c"
764
    )]
765
    #[case(
766
        "no chunked ctl chars here\x20\x09\x0b\x0c",
767
        "no chunked ctl chars here\x20\x09\x0b\x0c",
768
        ""
769
    )]
770
    #[case(
771
        "\x20\x09\x0b\x0cno chunked ctl chars here\x20\x09\x0b\x0c",
772
        "no chunked ctl chars here\x20\x09\x0b\x0c",
773
        "\x20\x09\x0b\x0c"
774
    )]
775
    fn test_take_chunked_ctl_chars(
776
        #[case] input: &str, #[case] remaining: &str, #[case] hex_digits: &str,
777
    ) {
778
        //(trailing non-LWS characters, found hex digits)
779
        assert_eq!(
780
            take_chunked_ctl_chars(input.as_bytes()).unwrap(),
781
            (remaining.as_bytes(), hex_digits.as_bytes())
782
        );
783
    }
784
785
    #[rstest]
786
    #[case("", true)]
787
    #[case("68656c6c6f", true)]
788
    #[case("\x0d\x0a\x20\x09\x0b\x0c68656c6c6f", true)]
789
    #[case("X5O!P%@AP", false)]
790
    #[case("\x0d\x0a\x20\x09\x0b\x0cX5O!P%@AP", false)]
791
    fn test_is_valid_chunked_length_data(#[case] input: &str, #[case] expected: bool) {
792
        assert_eq!(is_valid_chunked_length_data(input.as_bytes()), expected);
793
    }
794
795
    #[rstest]
796
    #[case("", false, true, ("", ""))]
797
    #[case("ONE TWO THREE", false, true, ("ONE", "TWO THREE"))]
798
    #[case("ONE TWO THREE", true, true, ("ONE TWO", "THREE"))]
799
    #[case("ONE   TWO   THREE", false, true, ("ONE", "TWO   THREE"))]
800
    #[case("ONE   TWO   THREE", true, true, ("ONE   TWO", "THREE"))]
801
    #[case("ONE", false, true, ("ONE", ""))]
802
    #[case("ONE", true, true, ("ONE", ""))]
803
    fn test_split_on_predicate(
804
        #[case] input: &str, #[case] reverse: bool, #[case] trim: bool,
805
        #[case] expected: (&str, &str),
806
    ) {
807
        assert_eq!(
808
            split_on_predicate(input.as_bytes(), reverse, trim, |c| *c == 0x20),
809
            (expected.0.as_bytes(), expected.1.as_bytes())
810
        );
811
    }
812
}