tinystr/int_ops.rs
1// This file is part of ICU4X. For terms of use, please see the file
2// called LICENSE at the top level of the ICU4X source tree
3// (online at: https://github.com/unicode-org/icu4x/blob/main/LICENSE ).
4
5use crate::asciibyte::AsciiByte;
6
7/// Internal helper struct that performs operations on aligned integers.
8/// Supports strings up to 4 bytes long.
9#[repr(transparent)]
10pub struct Aligned4(u32);
11
12impl Aligned4 {
13 /// # Panics
14 /// Panics if N is greater than 4
15 #[inline]
16 pub const fn from_ascii_bytes<const N: usize>(src: &[AsciiByte; N]) -> Self {
17 let mut bytes = [0; 4];
18 let mut i = 0;
19 // The function documentation defines when panics may occur
20 #[expect(clippy::indexing_slicing)]
21 while i < N {
22 bytes[i] = src[i] as u8;
23 i += 1;
24 }
25 Self(u32::from_ne_bytes(bytes))
26 }
27
28 pub const fn len(&self) -> usize {
29 let word = self.0;
30 #[cfg(target_endian = "little")]
31 let len = (4 - word.leading_zeros() / 8) as usize;
32 #[cfg(target_endian = "big")]
33 let len = (4 - word.trailing_zeros() / 8) as usize;
34 len
35 }
36
37 pub const fn is_ascii_alphabetic(&self) -> bool {
38 let word = self.0;
39 // Each of the following bitmasks set *the high bit* (0x8) to 0 for valid and 1 for invalid.
40 // `mask` sets all NUL bytes to 0.
41 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
42 // `lower` converts the string to lowercase. It may also change the value of non-alpha
43 // characters, but this does not matter for the alphabetic test that follows.
44 let lower = word | 0x2020_2020;
45 // `alpha` sets all alphabetic bytes to 0. We only need check for lowercase characters.
46 let alpha = !(lower + 0x1f1f_1f1f) | (lower + 0x0505_0505);
47 // The overall string is valid if every character passes at least one test.
48 // We performed two tests here: non-NUL (`mask`) and alphabetic (`alpha`).
49 (alpha & mask) == 0
50 }
51
52 pub const fn is_ascii_alphanumeric(&self) -> bool {
53 let word = self.0;
54 // See explanatory comments in is_ascii_alphabetic
55 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
56 let numeric = !(word + 0x5050_5050) | (word + 0x4646_4646);
57 let lower = word | 0x2020_2020;
58 let alpha = !(lower + 0x1f1f_1f1f) | (lower + 0x0505_0505);
59 (alpha & numeric & mask) == 0
60 }
61
62 pub const fn is_ascii_numeric(&self) -> bool {
63 let word = self.0;
64 // See explanatory comments in is_ascii_alphabetic
65 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
66 let numeric = !(word + 0x5050_5050) | (word + 0x4646_4646);
67 (numeric & mask) == 0
68 }
69
70 pub const fn is_ascii_lowercase(&self) -> bool {
71 let word = self.0;
72 // For efficiency, this function tests for an invalid string rather than a valid string.
73 // A string is ASCII lowercase iff it contains no uppercase ASCII characters.
74 // `invalid_case` sets all uppercase ASCII characters to 0 and all others to 1.
75 let invalid_case = !(word + 0x3f3f_3f3f) | (word + 0x2525_2525);
76 // The string is valid if it contains no invalid characters (if all high bits are 1).
77 (invalid_case & 0x8080_8080) == 0x8080_8080
78 }
79
80 pub const fn is_ascii_titlecase(&self) -> bool {
81 let word = self.0;
82 // See explanatory comments in is_ascii_lowercase
83 let invalid_case = if cfg!(target_endian = "little") {
84 !(word + 0x3f3f_3f1f) | (word + 0x2525_2505)
85 } else {
86 !(word + 0x1f3f_3f3f) | (word + 0x0525_2525)
87 };
88 (invalid_case & 0x8080_8080) == 0x8080_8080
89 }
90
91 pub const fn is_ascii_uppercase(&self) -> bool {
92 let word = self.0;
93 // See explanatory comments in is_ascii_lowercase
94 let invalid_case = !(word + 0x1f1f_1f1f) | (word + 0x0505_0505);
95 (invalid_case & 0x8080_8080) == 0x8080_8080
96 }
97
98 pub const fn is_ascii_alphabetic_lowercase(&self) -> bool {
99 let word = self.0;
100 // `mask` sets all NUL bytes to 0.
101 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
102 // `lower_alpha` sets all lowercase ASCII characters to 0 and all others to 1.
103 let lower_alpha = !(word + 0x1f1f_1f1f) | (word + 0x0505_0505);
104 // The overall string is valid if every character passes at least one test.
105 // We performed two tests here: non-NUL (`mask`) and lowercase ASCII character (`alpha`).
106 (lower_alpha & mask) == 0
107 }
108
109 pub const fn is_ascii_alphabetic_titlecase(&self) -> bool {
110 let word = self.0;
111 // See explanatory comments in is_ascii_alphabetic_lowercase
112 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
113 let title_case = if cfg!(target_endian = "little") {
114 !(word + 0x1f1f_1f3f) | (word + 0x0505_0525)
115 } else {
116 !(word + 0x3f1f_1f1f) | (word + 0x2505_0505)
117 };
118 (title_case & mask) == 0
119 }
120
121 pub const fn is_ascii_alphabetic_uppercase(&self) -> bool {
122 let word = self.0;
123 // See explanatory comments in is_ascii_alphabetic_lowercase
124 let mask = (word + 0x7f7f_7f7f) & 0x8080_8080;
125 let upper_alpha = !(word + 0x3f3f_3f3f) | (word + 0x2525_2525);
126 (upper_alpha & mask) == 0
127 }
128
129 pub const fn to_ascii_lowercase(&self) -> [AsciiByte; 4] {
130 let word = self.0;
131 let mask = ((word + 0x3f3f_3f3f) & !(word + 0x2525_2525) & 0x8080_8080) >> 2;
132 let result = word | mask;
133 // SAFETY: The existing word is ASCII, and it is bitwise-ORed with `mask`,
134 // a subset of 0x20_20_20_... (0x80_... >> 2), which keeps the word
135 // in the ASCII range (all bytes <= 127).
136 unsafe { AsciiByte::to_ascii_byte_array(&result.to_ne_bytes()) }
137 }
138
139 pub const fn to_ascii_titlecase(&self) -> [AsciiByte; 4] {
140 let word = self.0.to_le();
141 let mask = ((word + 0x3f3f_3f1f) & !(word + 0x2525_2505) & 0x8080_8080) >> 2;
142 let result = (word | mask) & !(0x20 & mask);
143 // SAFETY: The existing word is ASCII, and it is bitwise-ORed with `mask`, a subset of 0x20_20_20_...
144 // (0x80_... >> 2), which keeps the word in the ASCII range (all bytes <= 127). It is then ANDed
145 // which cannot introduce new bits.
146 unsafe { AsciiByte::to_ascii_byte_array(&u32::from_le(result).to_ne_bytes()) }
147 }
148
149 pub const fn to_ascii_uppercase(&self) -> [AsciiByte; 4] {
150 let word = self.0;
151 let mask = !(((word + 0x1f1f_1f1f) & !(word + 0x0505_0505) & 0x8080_8080) >> 2);
152 let result = word & mask;
153 // SAFETY: The existing word is ASCII, and it is bitwise-ANDed with `mask`,
154 // which cannot introduce new bits, so it stays ASCII.
155 unsafe { AsciiByte::to_ascii_byte_array(&result.to_ne_bytes()) }
156 }
157}
158
159/// Internal helper struct that performs operations on aligned integers.
160/// Supports strings up to 8 bytes long.
161#[repr(transparent)]
162pub struct Aligned8(u64);
163
164impl Aligned8 {
165 /// # Panics
166 /// Panics if N is greater than 8
167 #[inline]
168 pub const fn from_ascii_bytes<const N: usize>(src: &[AsciiByte; N]) -> Self {
169 let mut bytes = [0; 8];
170 let mut i = 0;
171 // The function documentation defines when panics may occur
172 #[expect(clippy::indexing_slicing)]
173 while i < N {
174 bytes[i] = src[i] as u8;
175 i += 1;
176 }
177 Self(u64::from_ne_bytes(bytes))
178 }
179
180 pub const fn len(&self) -> usize {
181 let word = self.0;
182 #[cfg(target_endian = "little")]
183 let len = (8 - word.leading_zeros() / 8) as usize;
184 #[cfg(target_endian = "big")]
185 let len = (8 - word.trailing_zeros() / 8) as usize;
186 len
187 }
188
189 pub const fn is_ascii_alphabetic(&self) -> bool {
190 let word = self.0;
191 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
192 let lower = word | 0x2020_2020_2020_2020;
193 let alpha = !(lower + 0x1f1f_1f1f_1f1f_1f1f) | (lower + 0x0505_0505_0505_0505);
194 (alpha & mask) == 0
195 }
196
197 pub const fn is_ascii_alphanumeric(&self) -> bool {
198 let word = self.0;
199 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
200 let numeric = !(word + 0x5050_5050_5050_5050) | (word + 0x4646_4646_4646_4646);
201 let lower = word | 0x2020_2020_2020_2020;
202 let alpha = !(lower + 0x1f1f_1f1f_1f1f_1f1f) | (lower + 0x0505_0505_0505_0505);
203 (alpha & numeric & mask) == 0
204 }
205
206 pub const fn is_ascii_numeric(&self) -> bool {
207 let word = self.0;
208 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
209 let numeric = !(word + 0x5050_5050_5050_5050) | (word + 0x4646_4646_4646_4646);
210 (numeric & mask) == 0
211 }
212
213 pub const fn is_ascii_lowercase(&self) -> bool {
214 let word = self.0;
215 let invalid_case = !(word + 0x3f3f_3f3f_3f3f_3f3f) | (word + 0x2525_2525_2525_2525);
216 (invalid_case & 0x8080_8080_8080_8080) == 0x8080_8080_8080_8080
217 }
218
219 pub const fn is_ascii_titlecase(&self) -> bool {
220 let word = self.0;
221 let invalid_case = if cfg!(target_endian = "little") {
222 !(word + 0x3f3f_3f3f_3f3f_3f1f) | (word + 0x2525_2525_2525_2505)
223 } else {
224 !(word + 0x1f3f_3f3f_3f3f_3f3f) | (word + 0x0525_2525_2525_2525)
225 };
226 (invalid_case & 0x8080_8080_8080_8080) == 0x8080_8080_8080_8080
227 }
228
229 pub const fn is_ascii_uppercase(&self) -> bool {
230 let word = self.0;
231 let invalid_case = !(word + 0x1f1f_1f1f_1f1f_1f1f) | (word + 0x0505_0505_0505_0505);
232 (invalid_case & 0x8080_8080_8080_8080) == 0x8080_8080_8080_8080
233 }
234
235 pub const fn is_ascii_alphabetic_lowercase(&self) -> bool {
236 let word = self.0;
237 // `mask` sets all NUL bytes to 0.
238 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
239 // `lower_alpha` sets all lowercase ASCII characters to 0 and all others to 1.
240 let lower_alpha = !(word + 0x1f1f_1f1f_1f1f_1f1f) | (word + 0x0505_0505_0505_0505);
241 // The overall string is valid if every character passes at least one test.
242 // We performed two tests here: non-NUL (`mask`) and lowercase ASCII character (`alpha`).
243 (lower_alpha & mask) == 0
244 }
245
246 pub const fn is_ascii_alphabetic_titlecase(&self) -> bool {
247 let word = self.0;
248 // See explanatory comments in is_ascii_alphabetic_lowercase
249 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
250 let title_case = if cfg!(target_endian = "little") {
251 !(word + 0x1f1f_1f1f_1f1f_1f3f) | (word + 0x0505_0505_0505_0525)
252 } else {
253 !(word + 0x3f1f_1f1f_1f1f_1f1f) | (word + 0x2505_0505_0505_0505)
254 };
255 (title_case & mask) == 0
256 }
257
258 pub const fn is_ascii_alphabetic_uppercase(&self) -> bool {
259 let word = self.0;
260 // See explanatory comments in is_ascii_alphabetic_lowercase
261 let mask = (word + 0x7f7f_7f7f_7f7f_7f7f) & 0x8080_8080_8080_8080;
262 let upper_alpha = !(word + 0x3f3f_3f3f_3f3f_3f3f) | (word + 0x2525_2525_2525_2525);
263 (upper_alpha & mask) == 0
264 }
265
266 pub const fn to_ascii_lowercase(&self) -> [AsciiByte; 8] {
267 let word = self.0;
268 let mask = ((word + 0x3f3f_3f3f_3f3f_3f3f)
269 & !(word + 0x2525_2525_2525_2525)
270 & 0x8080_8080_8080_8080)
271 >> 2;
272 let result = word | mask;
273 // SAFETY: The existing word is ASCII, and it is bitwise-ORed with `mask`,
274 // a subset of 0x20_20_20_... (0x80_... >> 2), which keeps the word
275 // in the ASCII range (all bytes <= 127).
276 unsafe { AsciiByte::to_ascii_byte_array(&result.to_ne_bytes()) }
277 }
278
279 pub const fn to_ascii_titlecase(&self) -> [AsciiByte; 8] {
280 let word = self.0.to_le();
281 let mask = ((word + 0x3f3f_3f3f_3f3f_3f1f)
282 & !(word + 0x2525_2525_2525_2505)
283 & 0x8080_8080_8080_8080)
284 >> 2;
285 let result = (word | mask) & !(0x20 & mask);
286 // SAFETY: The existing word is ASCII, and it is bitwise-ORed with `mask`, a subset of 0x20_20_20_...
287 // (0x80_... >> 2), which keeps the word in the ASCII range (all bytes <= 127). It is then ANDed
288 // which cannot introduce new bits.
289 unsafe { AsciiByte::to_ascii_byte_array(&u64::from_le(result).to_ne_bytes()) }
290 }
291
292 pub const fn to_ascii_uppercase(&self) -> [AsciiByte; 8] {
293 let word = self.0;
294 let mask = !(((word + 0x1f1f_1f1f_1f1f_1f1f)
295 & !(word + 0x0505_0505_0505_0505)
296 & 0x8080_8080_8080_8080)
297 >> 2);
298 let result = word & mask;
299 // SAFETY: The existing word is ASCII, and it is bitwise-ANDed with `mask`,
300 // which cannot introduce new bits, so it stays ASCII.
301 unsafe { AsciiByte::to_ascii_byte_array(&result.to_ne_bytes()) }
302 }
303}