/rust/registry/src/index.crates.io-1949cf8c6b5b557f/potential_utf-0.1.6/src/ustr.rs
Line | Count | Source |
1 | | // This file is part of ICU4X. For terms of use, please see the file |
2 | | // called LICENSE at the top level of the ICU4X source tree |
3 | | // (online at: https://github.com/unicode-org/icu4x/blob/main/LICENSE ). |
4 | | |
5 | | #[cfg(feature = "alloc")] |
6 | | use alloc::boxed::Box; |
7 | | use core::cmp::Ordering; |
8 | | use core::fmt; |
9 | | use core::ops::Deref; |
10 | | |
11 | | /// A byte slice that is expected to be a UTF-8 string but does not enforce that invariant. |
12 | | /// |
13 | | /// Use this type instead of `str` if you don't need to enforce UTF-8 during deserialization. For |
14 | | /// example, strings that are keys of a map don't need to ever be reified as `str`s. |
15 | | /// |
16 | | /// [`PotentialUtf8`] derefs to `[u8]`. To obtain a `str`, use [`Self::try_as_str()`]. |
17 | | /// |
18 | | /// The main advantage of this type over `[u8]` is that it serializes as a string in |
19 | | /// human-readable formats like JSON. |
20 | | /// |
21 | | /// # Examples |
22 | | /// |
23 | | /// Using an [`PotentialUtf8`] as the key of a [`ZeroMap`]: |
24 | | /// |
25 | | /// ``` |
26 | | /// use potential_utf::PotentialUtf8; |
27 | | /// use zerovec::ZeroMap; |
28 | | /// |
29 | | /// // This map is cheap to deserialize, as we don't need to perform UTF-8 validation. |
30 | | /// let map: ZeroMap<PotentialUtf8, u8> = [ |
31 | | /// (PotentialUtf8::from_bytes(b"abc"), 11), |
32 | | /// (PotentialUtf8::from_bytes(b"def"), 22), |
33 | | /// (PotentialUtf8::from_bytes(b"ghi"), 33), |
34 | | /// ] |
35 | | /// .into_iter() |
36 | | /// .collect(); |
37 | | /// |
38 | | /// let key = "abc"; |
39 | | /// let value = map.get_copied(PotentialUtf8::from_str(key)); |
40 | | /// assert_eq!(Some(11), value); |
41 | | /// ``` |
42 | | /// |
43 | | /// [`ZeroMap`]: zerovec::ZeroMap |
44 | | #[repr(transparent)] |
45 | | #[derive(PartialEq, Eq, PartialOrd, Ord)] |
46 | | #[allow(clippy::exhaustive_structs)] // transparent newtype |
47 | | pub struct PotentialUtf8(pub [u8]); |
48 | | |
49 | | impl fmt::Debug for PotentialUtf8 { |
50 | 0 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
51 | | // Debug as a string if possible |
52 | 0 | match self.try_as_str() { |
53 | 0 | Ok(s) => fmt::Debug::fmt(s, f), |
54 | 0 | Err(_) => fmt::Debug::fmt(&self.0, f), |
55 | | } |
56 | 0 | } |
57 | | } |
58 | | |
59 | | impl PotentialUtf8 { |
60 | | /// Create a [`PotentialUtf8`] from a byte slice. |
61 | | #[inline] |
62 | 0 | pub const fn from_bytes(other: &[u8]) -> &Self { |
63 | | // Safety: PotentialUtf8 is transparent over [u8] |
64 | 0 | unsafe { &*(other as *const [u8] as *const Self) } |
65 | 0 | } |
66 | | |
67 | | /// Create a [`PotentialUtf8`] from a string slice. |
68 | | #[inline] |
69 | 0 | pub const fn from_str(s: &str) -> &Self { |
70 | 0 | Self::from_bytes(s.as_bytes()) |
71 | 0 | } |
72 | | |
73 | | /// Create a [`PotentialUtf8`] from boxed bytes. |
74 | | /// |
75 | | /// ✨ *Enabled with the `alloc` Cargo feature.* |
76 | | #[inline] |
77 | | #[cfg(feature = "alloc")] |
78 | | pub fn from_boxed_bytes(other: Box<[u8]>) -> Box<Self> { |
79 | | // Safety: PotentialUtf8 is transparent over [u8] therefore |
80 | | // the cast between [u8] and Self is sound. |
81 | | // Box::into_raw returns a well aligned pointer that is not |
82 | | // null. The cast then changes the type, but the pointer |
83 | | // is still well aligned, not null and created by the |
84 | | // same global allocator, so it's safe to use Box::from_raw |
85 | | // to create a new Box instance out of it. |
86 | | // |
87 | | // We cannot directly transmute the `Box` here as |
88 | | // there is no gurantee about the layout of `Box` with |
89 | | // unsized types. |
90 | | unsafe { Box::from_raw(Box::into_raw(other) as *mut Self) } |
91 | | } |
92 | | |
93 | | /// Create a [`PotentialUtf8`] from a boxed `str`. |
94 | | /// |
95 | | /// ✨ *Enabled with the `alloc` Cargo feature.* |
96 | | #[inline] |
97 | | #[cfg(feature = "alloc")] |
98 | | pub fn from_boxed_str(other: Box<str>) -> Box<Self> { |
99 | | Self::from_boxed_bytes(other.into_boxed_bytes()) |
100 | | } |
101 | | |
102 | | /// Get the bytes from a [`PotentialUtf8`]. |
103 | | #[inline] |
104 | 0 | pub const fn as_bytes(&self) -> &[u8] { |
105 | 0 | &self.0 |
106 | 0 | } |
107 | | |
108 | | /// Attempt to convert a [`PotentialUtf8`] to a `str`. |
109 | | /// |
110 | | /// # Examples |
111 | | /// |
112 | | /// ``` |
113 | | /// use potential_utf::PotentialUtf8; |
114 | | /// |
115 | | /// static A: &PotentialUtf8 = PotentialUtf8::from_bytes(b"abc"); |
116 | | /// |
117 | | /// let b = A.try_as_str().unwrap(); |
118 | | /// assert_eq!(b, "abc"); |
119 | | /// ``` |
120 | | // Note: this is const starting in 1.63 |
121 | | #[inline] |
122 | 0 | pub fn try_as_str(&self) -> Result<&str, core::str::Utf8Error> { |
123 | 0 | core::str::from_utf8(&self.0) |
124 | 0 | } |
125 | | } |
126 | | |
127 | | impl<'a> From<&'a str> for &'a PotentialUtf8 { |
128 | | #[inline] |
129 | 0 | fn from(other: &'a str) -> Self { |
130 | 0 | PotentialUtf8::from_str(other) |
131 | 0 | } |
132 | | } |
133 | | |
134 | | impl PartialEq<str> for PotentialUtf8 { |
135 | 0 | fn eq(&self, other: &str) -> bool { |
136 | 0 | self.eq(Self::from_str(other)) |
137 | 0 | } |
138 | | } |
139 | | |
140 | | impl PartialOrd<str> for PotentialUtf8 { |
141 | 0 | fn partial_cmp(&self, other: &str) -> Option<Ordering> { |
142 | 0 | self.partial_cmp(Self::from_str(other)) |
143 | 0 | } |
144 | | } |
145 | | |
146 | | impl PartialEq<PotentialUtf8> for str { |
147 | 0 | fn eq(&self, other: &PotentialUtf8) -> bool { |
148 | 0 | PotentialUtf8::from_str(self).eq(other) |
149 | 0 | } |
150 | | } |
151 | | |
152 | | impl PartialOrd<PotentialUtf8> for str { |
153 | 0 | fn partial_cmp(&self, other: &PotentialUtf8) -> Option<Ordering> { |
154 | 0 | PotentialUtf8::from_str(self).partial_cmp(other) |
155 | 0 | } |
156 | | } |
157 | | |
158 | | #[cfg(feature = "alloc")] |
159 | | impl From<Box<str>> for Box<PotentialUtf8> { |
160 | | #[inline] |
161 | | fn from(other: Box<str>) -> Self { |
162 | | PotentialUtf8::from_boxed_str(other) |
163 | | } |
164 | | } |
165 | | |
166 | | impl Deref for PotentialUtf8 { |
167 | | type Target = [u8]; |
168 | 0 | fn deref(&self) -> &Self::Target { |
169 | 0 | &self.0 |
170 | 0 | } |
171 | | } |
172 | | |
173 | | /// This impl requires enabling the optional `zerovec` Cargo feature |
174 | | #[cfg(all(feature = "zerovec", feature = "alloc"))] |
175 | | impl<'a> zerovec::maps::ZeroMapKV<'a> for PotentialUtf8 { |
176 | | type Container = zerovec::VarZeroVec<'a, PotentialUtf8>; |
177 | | type Slice = zerovec::VarZeroSlice<PotentialUtf8>; |
178 | | type GetType = PotentialUtf8; |
179 | | type OwnedType = Box<PotentialUtf8>; |
180 | | } |
181 | | |
182 | | // Safety (based on the safety checklist on the VarULE trait): |
183 | | // 1. PotentialUtf8 does not include any uninitialized or padding bytes (transparent over a ULE) |
184 | | // 2. PotentialUtf8 is aligned to 1 byte (transparent over a ULE) |
185 | | // 3. The impl of `validate_bytes()` returns an error if any byte is not valid (impossible) |
186 | | // 4. The impl of `validate_bytes()` returns an error if the slice cannot be used in its entirety (impossible) |
187 | | // 5. The impl of `from_bytes_unchecked()` returns a reference to the same data (returns the argument directly) |
188 | | // 6. All other methods are defaulted |
189 | | // 7. `[T]` byte equality is semantic equality (transparent over a ULE) |
190 | | /// This impl requires enabling the optional `zerovec` Cargo feature |
191 | | #[cfg(feature = "zerovec")] |
192 | | unsafe impl zerovec::ule::VarULE for PotentialUtf8 { |
193 | | #[inline] |
194 | 0 | fn validate_bytes(_: &[u8]) -> Result<(), zerovec::ule::UleError> { |
195 | 0 | Ok(()) |
196 | 0 | } |
197 | | #[inline] |
198 | 0 | unsafe fn from_bytes_unchecked(bytes: &[u8]) -> &Self { |
199 | 0 | PotentialUtf8::from_bytes(bytes) |
200 | 0 | } |
201 | | } |
202 | | |
203 | | /// This impl requires enabling the optional `serde` Cargo feature |
204 | | #[cfg(feature = "serde")] |
205 | | impl serde_core::Serialize for PotentialUtf8 { |
206 | | fn serialize<S>(&self, serializer: S) -> Result<S::Ok, S::Error> |
207 | | where |
208 | | S: serde_core::Serializer, |
209 | | { |
210 | | use serde_core::ser::Error; |
211 | | let s = self |
212 | | .try_as_str() |
213 | | .map_err(|_| S::Error::custom("invalid UTF-8 in PotentialUtf8"))?; |
214 | | if serializer.is_human_readable() { |
215 | | serializer.serialize_str(s) |
216 | | } else { |
217 | | serializer.serialize_bytes(s.as_bytes()) |
218 | | } |
219 | | } |
220 | | } |
221 | | |
222 | | /// This impl requires enabling the optional `serde` Cargo feature |
223 | | #[cfg(all(feature = "serde", feature = "alloc"))] |
224 | | impl<'de> serde_core::Deserialize<'de> for Box<PotentialUtf8> { |
225 | | fn deserialize<D>(deserializer: D) -> Result<Self, D::Error> |
226 | | where |
227 | | D: serde_core::Deserializer<'de>, |
228 | | { |
229 | | if deserializer.is_human_readable() { |
230 | | let boxed_str = Box::<str>::deserialize(deserializer)?; |
231 | | Ok(PotentialUtf8::from_boxed_str(boxed_str)) |
232 | | } else { |
233 | | let boxed_bytes = Box::<[u8]>::deserialize(deserializer)?; |
234 | | Ok(PotentialUtf8::from_boxed_bytes(boxed_bytes)) |
235 | | } |
236 | | } |
237 | | } |
238 | | |
239 | | /// This impl requires enabling the optional `serde` Cargo feature |
240 | | #[cfg(feature = "serde")] |
241 | | impl<'de, 'a> serde_core::Deserialize<'de> for &'a PotentialUtf8 |
242 | | where |
243 | | 'de: 'a, |
244 | | { |
245 | | fn deserialize<D>(deserializer: D) -> Result<Self, D::Error> |
246 | | where |
247 | | D: serde_core::Deserializer<'de>, |
248 | | { |
249 | | if deserializer.is_human_readable() { |
250 | | let s = <&str>::deserialize(deserializer)?; |
251 | | Ok(PotentialUtf8::from_str(s)) |
252 | | } else { |
253 | | let bytes = <&[u8]>::deserialize(deserializer)?; |
254 | | Ok(PotentialUtf8::from_bytes(bytes)) |
255 | | } |
256 | | } |
257 | | } |
258 | | |
259 | | /// A `u16` slice that is expected to be a UTF-16 string but does not enforce that invariant. |
260 | | /// |
261 | | /// See [`PotentialUtf8`] for more info. |
262 | | #[repr(transparent)] |
263 | | #[derive(PartialEq, Eq, PartialOrd, Ord)] |
264 | | #[allow(clippy::exhaustive_structs)] // transparent newtype |
265 | | pub struct PotentialUtf16(pub [u16]); |
266 | | |
267 | | impl fmt::Debug for PotentialUtf16 { |
268 | 0 | fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { |
269 | | // Debug as a string if possible |
270 | 0 | for c in char::decode_utf16(self.0.iter().copied()) { |
271 | 0 | match c { |
272 | 0 | Ok(c) => write!(f, "{c}")?, |
273 | 0 | Err(e) => write!(f, "\\0x{:x}", e.unpaired_surrogate())?, |
274 | | } |
275 | | } |
276 | 0 | Ok(()) |
277 | 0 | } |
278 | | } |
279 | | |
280 | | impl PotentialUtf16 { |
281 | | /// Create a [`PotentialUtf16`] from a u16 slice. |
282 | | #[inline] |
283 | 0 | pub const fn from_slice(other: &[u16]) -> &Self { |
284 | | // Safety: PotentialUtf16 is transparent over [u16] |
285 | 0 | unsafe { &*(other as *const [u16] as *const Self) } |
286 | 0 | } |
287 | | |
288 | | /// Iterates the characters of the string. |
289 | | /// |
290 | | /// Returns [`char::REPLACEMENT_CHARACTER`] if invalid surrogates are encountered. |
291 | 0 | pub fn chars(&self) -> impl Iterator<Item = char> + '_ { |
292 | 0 | char::decode_utf16(self.0.iter().copied()).map(|c| c.unwrap_or(char::REPLACEMENT_CHARACTER)) |
293 | 0 | } |
294 | | } |