Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_data/tests/base2x.rs

28.9 KiB, 14 runs

created by r1870400018:281, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1use oxedyne_fe2o3_core::{
2 prelude::*,
3 test::test_it,
4};
5use oxedyne_fe2o3_text::{
6 base2x::{
7 self,
8 Base2x,
9 },
10 string::Stringer,
11};
12
13use base64::{
14 prelude::BASE64_STANDARD,
15 Engine,
16};
17
18
19// Default constants.
20const X_LEN: usize = 3;
21const A_LEN: usize = base2x::alphabet_size(X_LEN as u32);
22
23pub fn test_base2x(filter: &'static str) -> Outcome<()> {
24
25 res!(test_it(filter, &["Simple create 000", "all", "base2x"], || {
26 const A_LEN: usize = 9;
27 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc123"));
28 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
29 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
30 Ok(_base2x) => return Err(err!(
31 "Alphabet of length {} should be invalid.", alphabet.len();
32 Invalid, Input, Test)),
33 Err(e) => msg!("Correctly triggered: {}", e),
34 };
35 Ok(())
36 }));
37
38 res!(test_it(filter, &["Simple create 010", "all", "base2x"], || {
39 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc12"));
40 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
41 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
42 Ok(_base2x) => (),
43 Err(e) => return Err(e),
44 };
45 Ok(())
46 }));
47
48 res!(test_it(filter, &["Simple create 020", "all", "base2x"], || {
49 const A_LEN: usize = 300;
50 let repeated_char = 'A';
51 let repeat_count = 300;
52 let alphabet: String =
53 std::iter::repeat(repeated_char).take(repeat_count).collect();
54
55 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet(&alphabet));
56 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
57 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
58 Ok(_base2x) => return Err(err!(
59 "Alphabet of length {} should be invalid.", alphabet.len();
60 Invalid, Input, Test)),
61 Err(e) => test!("Correctly triggered: {}", e),
62 };
63 Ok(())
64 }));
65
66 res!(test_it(filter, &["Simple create 030", "all", "base2x", "unicode"], || {
67 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCab\u{e9}12"));
68 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("123"));
69 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
70 Ok(_base2x) => (),
71 Err(e) => return Err(e),
72 };
73 Ok(())
74 }));
75
76 res!(test_it(filter, &["Round trip 000", "all", "base2x", "hematite"], || {
77 let base2x = base2x::HEMATITE64;
78 let a = base2x.alphabet_size();
79 let x = base2x.token_size();
80 let input = fmt!("This Hematite alphabet has {} characters", a);
81 let input_byts = input.as_bytes();
82 trace!("This is the alphabet we'll be using:");
83 for line in base2x.fmt_char_map() {
84 trace!(" {}", line);
85 }
86 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
87 for byt in input_byts {
88 debug!(" {:08b}", byt);
89 }
90 test!("Each of the {} tokens has a length of {} bits.", a, x);
91 test!("Performing Base2x encoding from bytes to string...");
92 let encoded = base2x.to_string(&input_byts);
93 test!("Base2x encoding: '{}'", encoded);
94
95 test!("Performing Base2x decoding from string to bytes...");
96 let byts = res!(base2x.from_str(&encoded));
97 let mut s = String::new();
98 debug!("This gives us the following bytes:");
99 for byt in &byts {
100 debug!(" {:08b}", byt);
101 s.push_str(&fmt!("{:08b}", byt));
102 }
103 let g = Stringer::new(s);
104 debug!("As a bit string: {}", g.insert_every("_", 8));
105 debug!("As a bit string: {}", g.insert_every("_", x));
106
107 let decoded = res!(std::str::from_utf8(&byts));
108 test!("When the bytes are converted back to a string we get: '{}'", decoded);
109 req!(decoded, input);
110 Ok(())
111 }));
112
113 res!(test_it(filter, &["Round trip 005", "all", "base2x", "hematite"], || {
114 let base2x = base2x::HEMATITE32;
115 let a = base2x.alphabet_size();
116 let x = base2x.token_size();
117 let input = fmt!("This Hematite alphabet has {} characters", a);
118 let input_byts = input.as_bytes();
119 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
120 for byt in input_byts {
121 debug!(" {:08b}", byt);
122 }
123 test!("Each of the {} tokens has a length of {} bits.", a, x);
124 test!("Performing Base2x encoding from bytes to string...");
125 let encoded = base2x.to_string(&input_byts);
126 test!("Base2x encoding: '{}'", encoded);
127
128 test!("Performing Base2x decoding from string to bytes...");
129 debug!("This gives us the following bytes:");
130 let byts = res!(base2x.from_str(&encoded));
131 let mut s = String::new();
132 for byt in &byts {
133 debug!(" {:08b}", byt);
134 s.push_str(&fmt!("{:08b}", byt));
135 }
136 let g = Stringer::new(s);
137 debug!("As a bit string: {}", g.insert_every("_", 8));
138 debug!("As a bit string: {}", g.insert_every("_", x));
139
140 let decoded = res!(std::str::from_utf8(&byts));
141 test!("When the bytes are converted back to a string we get: '{}'", decoded);
142 req!(decoded, input);
143 Ok(())
144 }));
145
146 res!(test_it(filter, &["Round trip 010", "all", "base2x"], || {
147 let base2x = base2x::BASE64;
148 let a = base2x.alphabet_size();
149 let x = base2x.token_size();
150 for line in base2x.fmt_char_map() {
151 debug!("{}", line);
152 }
153
154 let input = "Man";
155 test!("This is an example from the Wikipedia page for Base64.");
156 let input_byts = input.as_bytes();
157 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
158 for byt in input_byts {
159 test!(" {:08b}", byt);
160 }
161 test!("Each of the {} tokens has a length of {} bits.", a, x);
162 test!("Performing Base2x encoding from bytes to string...");
163 let encoded = base2x.to_string(&input_byts);
164 test!("Base2x encoding: '{}'", encoded);
165
166 let base64_encoded = BASE64_STANDARD.encode(&input_byts);
167 test!("Now let's encode using the standard alphabet of Base64: '{}'", base64_encoded);
168 test!("They should be identical, because there is no padding in this case.");
169 req!(encoded, base64_encoded);
170 test!("The character token mapping is:");
171 for c in encoded.chars() {
172 let token_opt = base2x.get_token(c);
173 test!(" '{}' -> {}", c,
174 token_opt.map_or(
175 fmt!("None"),
176 |b| fmt!("{:0width$b}", b, width = x),
177 ),
178 );
179 }
180
181 test!("Performing Base2x decoding from string to bytes...");
182 let byts = res!(base2x.from_str(&encoded));
183 test!("This gives us the following bytes:");
184 let mut s = String::new();
185 for byt in &byts {
186 test!(" {:08b}", byt);
187 s.push_str(&fmt!("{:08b}", byt));
188 }
189 let g = Stringer::new(s);
190 test!("As a bit string: {}", g.insert_every("_", 8));
191 test!("As a bit string: {}", g.insert_every("_", x));
192
193 test!("Performing Base2x encoding back from bytes to string...");
194 let decoded = res!(std::str::from_utf8(&byts));
195 test!("When the bytes are converted back to a string we get: '{}'", decoded);
196 req!(decoded, input);
197 Ok(())
198 }));
199
200 res!(test_it(filter, &["Round trip 011", "all", "base2x"], || {
201 let base2x = base2x::BASE64;
202 let a = base2x.alphabet_size();
203 let x = base2x.token_size();
204 for line in base2x.fmt_char_map() {
205 debug!("{}", line);
206 }
207
208 let input = "Ma";
209 test!("This is an example from the Wikipedia page for Base64.");
210 let input_byts = input.as_bytes();
211 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
212 for byt in input_byts {
213 test!(" {:08b}", byt);
214 }
215 test!("Each of the {} tokens has a length of {} bits.", a, x);
216 test!("Performing Base2x encoding from bytes to string...");
217 let encoded = base2x.to_string(&input_byts);
218 test!("Base2x encoding: '{}'", encoded);
219 let mut compare = encoded.clone();
220 compare.pop();
221 test!("This should include padding, so let's remove the Base2x-specific part: '{}'", compare);
222 let base64_encoded = BASE64_STANDARD.encode(&input_byts);
223 test!("The character token mapping is:");
224 for c in encoded.chars() {
225 let token_opt = base2x.get_token(c);
226 test!(" '{}' -> {}", c,
227 token_opt.map_or(
228 fmt!("None"),
229 |b| fmt!("{:0width$b}", b, width = x),
230 ),
231 );
232 }
233 test!("Now let's encode using the standard alphabet of Base64: '{}'", base64_encoded);
234 test!("They should be identical, because they use a similar padding scheme.");
235 req!(compare, base64_encoded);
236
237 test!("Performing Base2x decoding from string to bytes...");
238 let byts = res!(base2x.from_str(&encoded));
239 test!("This gives us the following bytes:");
240 let mut s = String::new();
241 for byt in &byts {
242 test!(" {:08b}", byt);
243 s.push_str(&fmt!("{:08b}", byt));
244 }
245 let g = Stringer::new(s);
246 test!("As a bit string: {}", g.insert_every("_", 8));
247 test!("As a bit string: {}", g.insert_every("_", x));
248 test!("You can see at the end here that the last character is partial, with only 4 bits.");
249 test!("The padding appends 2 zero bits, giving 000100 or 'E'.");
250
251 test!("Performing Base2x encoding back from bytes to string...");
252 let decoded = res!(std::str::from_utf8(&byts));
253 test!("When the bytes are converted back to a string we get: '{}'", decoded);
254 req!(decoded, input);
255 Ok(())
256 }));
257
258 res!(test_it(filter, &["Round trip 100", "all", "base2x"], || {
259 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc12"));
260 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
261 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
262 test!("This is the alphabet we'll be using:");
263 for line in base2x.fmt_char_map() {
264 test!(" {}", line);
265 }
266 let mut input = "BA2c1baBCaa21b111AcCb".to_string();
267 test!("Let's start with an arbitrary Base2x encoding: '{}'", input);
268 test!("Since each token is {} bits long, that's {} bits which is not divisible by 8.",
269 X_LEN, input.len() * X_LEN);
270 test!("Which means padding is needed, so let's apply it using Base2x normalisation...");
271 let padding = base2x.normalise(&mut input);
272 if padding > 0 {
273 base2x.push_pad(&mut input, padding);
274 }
275 test!("Normalised: '{}' with {} bits of padding.", input, padding);
276 test!("The token values for each encoded character are:");
277 for c in input.chars() {
278 let token_opt = base2x.get_token(c);
279 test!(" '{}' -> {}", c,
280 token_opt.map_or(
281 fmt!("None"),
282 |b| fmt!("{:0width$b}", b, width = X_LEN),
283 ),
284 );
285 }
286 test!("Performing Base2x decoding from string to bytes...");
287 let byts = res!(base2x.from_str(&input));
288 test!("This gives us the following bytes:");
289 let mut s = String::new();
290 for byt in &byts {
291 test!(" {:08b}", byt);
292 s.push_str(&fmt!("{:08b}", byt));
293 }
294 let g = Stringer::new(s);
295 test!("As a bit string: {}", g.insert_every("_", 8));
296 test!("As a bit string: {}", g.insert_every("_", X_LEN));
297
298 test!("Performing Base2x encoding back from bytes to string...");
299 let encoded = base2x.to_string(&byts);
300 test!("When the bytes are Base2x encoded back to a string we get: '{}'", encoded);
301 req!(encoded, input);
302 Ok(())
303 }));
304
305 res!(test_it(filter, &["Round trip 200", "all", "base2x"], || {
306 let byts = vec![0b01001011, 0b11101000, 0b10110100, 0b01101111];
307 test!("input bytes:");
308 for byt in &byts {
309 test!("{:08b}", byt);
310 }
311 let alphabet = "ABCabc12";
312 test!("alphabet = '{}'", alphabet);
313 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet(alphabet));
314 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
315 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
316 let n = byts.len();
317 let n_bits = 8*n;
318 let n_tokens = n_bits / base2x.token_size();
319 let rem = n_bits - n_tokens * base2x.token_size();
320 test!("{} input bytes have {} bits, requring {} tokens with a remainder of {}",
321 n, n_bits, n_tokens, rem);
322 let mut s = String::new();
323 for byt in &byts {
324 s.push_str(&fmt!("{:08b}", byt));
325 }
326 let g = Stringer::new(s);
327 test!("bit string: {}", g.insert_every("_", 8));
328 test!("bit string: {}", g.insert_every("_", 3));
329 let encoded = base2x.to_string(&byts);
330 test!("encoded = '{}'", encoded);
331
332 test!("encoded '{}' has {} characters, requring {} bits",
333 encoded, encoded.len(), base2x.token_size() * encoded.len());
334 for c in encoded.chars() {
335 let token_opt = base2x.get_token(c);
336 test!(" '{}' -> {}", c,
337 token_opt.map_or(
338 fmt!("None"),
339 |b| fmt!("{:0width$b}", b, width = base2x.token_size()),
340 ),
341 );
342 }
343 // Now decode string -> bytes
344 let byts2 = res!(base2x.from_str(&encoded));
345 let mut s = String::new();
346 for byt in &byts2 {
347 s.push_str(&fmt!("{:08b}", byt));
348 }
349 let g = Stringer::new(s);
350 test!("decoded bit string: {}", g.insert_every("_", 8));
351 test!("decoded bit string: {}", g.insert_every("_", base2x.token_size()));
352 req!(byts2, byts);
353 Ok(())
354 }));
355
356 res!(test_it(filter, &["Round trip 300", "all", "base2x", "unicode"], || {
357 const X_LEN: usize = 9;
358 const A_LEN: usize = base2x::alphabet_size(X_LEN as u32);
359 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12345678_"));
360 let alphabet = [
361 '\u{1f600}',
362 '\u{1f976}',
363 '\u{1f4a9}',
364 '\u{1f63b}',
365 '\u{1f44c}',
366 '\u{1f441}',
367 '\u{1f937}',
368 '\u{1f3cb}',
369 '\u{1f984}',
370 '\u{1f30f}',
371 '\u{26f5}',
372 '\u{2708}',
373 '\u{1fa90}',
374 '\u{1fabf}',
375 '\u{1f438}',
376 '\u{1f40a}',
377 '\u{1f422}',
378 '\u{1f98e}',
379 '\u{1f40d}',
380 '\u{1f432}',
381 '\u{1f409}',
382 '\u{1f995}',
383 '\u{1f996}',
384 '\u{1f433}',
385 '\u{1f40b}',
386 '\u{1f42c}',
387 '\u{1f9ad}',
388 '\u{1f41f}',
389 '\u{1f420}',
390 '\u{1f421}',
391 '\u{1f988}',
392 '\u{1f419}',
393 '\u{1f41a}',
394 '\u{1fab8}',
395 '\u{1fabc}',
396 '\u{1f40c}',
397 '\u{1f98b}',
398 '\u{1f41b}',
399 '\u{1f41c}',
400 '\u{1f41d}',
401 '\u{1fab2}',
402 '\u{1f41e}',
403 '\u{1f997}',
404 '\u{1fab3}',
405 '\u{1f577}',
406 '\u{1f578}',
407 '\u{1f982}',
408 '\u{1f99f}',
409 '\u{1fab0}',
410 '\u{1fab1}',
411 '\u{1f9a0}',
412 '\u{1f490}',
413 '\u{1f338}',
414 '\u{1f4ae}',
415 '\u{1fab7}',
416 '\u{1f3f5}',
417 '\u{1f339}',
418 '\u{1f940}',
419 '\u{1f33a}',
420 '\u{1f33b}',
421 '\u{1f33c}',
422 '\u{1f337}',
423 '\u{1fabb}',
424 '\u{1f331}',
425 '\u{1fab4}',
426 '\u{1f332}',
427 '\u{1f333}',
428 '\u{1f334}',
429 '\u{1f335}',
430 '\u{1f33e}',
431 '\u{1f33f}',
432 '\u{2618}',
433 '\u{1f340}',
434 '\u{1f341}',
435 '\u{1f342}',
436 '\u{1f343}',
437 '\u{1fab9}',
438 '\u{1faba}',
439 '\u{1f344}',
440 '\u{1f347}',
441 '\u{1f348}',
442 '\u{1f349}',
443 '\u{1f34a}',
444 '\u{1f34b}',
445 '\u{1f34c}',
446 '\u{1f34d}',
447 '\u{1f96d}',
448 '\u{1f34e}',
449 '\u{1f34f}',
450 '\u{1f350}',
451 '\u{1f351}',
452 '\u{1f352}',
453 '\u{1f353}',
454 '\u{1fad0}',
455 '\u{1f95d}',
456 '\u{1f345}',
457 '\u{1fad2}',
458 '\u{1f965}',
459 '\u{1f951}',
460 '\u{1f346}',
461 '\u{1f954}',
462 '\u{1f955}',
463 '\u{1f33d}',
464 '\u{1f336}',
465 '\u{1fad1}',
466 '\u{1f952}',
467 '\u{1f96c}',
468 '\u{1f966}',
469 '\u{1f9c4}',
470 '\u{1f9c5}',
471 '\u{1f95c}',
472 '\u{1fad8}',
473 '\u{1f330}',
474 '\u{1fada}',
475 '\u{1fadb}',
476 '\u{1f35e}',
477 '\u{1f950}',
478 '\u{1f956}',
479 '\u{1fad3}',
480 '\u{1f968}',
481 '\u{1f96f}',
482 '\u{1f95e}',
483 '\u{1f9c7}',
484 '\u{1f9c0}',
485 '\u{1f356}',
486 '\u{1f357}',
487 '\u{1f969}',
488 '\u{1f953}',
489 '\u{1f354}',
490 '\u{1f35f}',
491 '\u{1f355}',
492 '\u{1f32d}',
493 '\u{1f96a}',
494 '\u{1f32e}',
495 '\u{1f32f}',
496 '\u{1fad4}',
497 '\u{1f959}',
498 '\u{1f9c6}',
499 '\u{1f95a}',
500 '\u{1f373}',
501 '\u{1f958}',
502 '\u{1f372}',
503 '\u{1fad5}',
504 '\u{1f963}',
505 '\u{1f957}',
506 '\u{1f37f}',
507 '\u{1f9c8}',
508 '\u{1f9c2}',
509 '\u{1f96b}',
510 '\u{1f371}',
511 '\u{1f358}',
512 '\u{1f359}',
513 '\u{1f35a}',
514 '\u{1f35b}',
515 '\u{1f35c}',
516 '\u{1f35d}',
517 '\u{1f360}',
518 '\u{1f362}',
519 '\u{1f363}',
520 '\u{1f364}',
521 '\u{1f365}',
522 '\u{1f96e}',
523 '\u{1f361}',
524 '\u{1f95f}',
525 '\u{1f960}',
526 '\u{1f961}',
527 '\u{1f980}',
528 '\u{1f99e}',
529 '\u{1f990}',
530 '\u{1f991}',
531 '\u{1f9aa}',
532 '\u{1f366}',
533 '\u{1f367}',
534 '\u{1f368}',
535 '\u{1f369}',
536 '\u{1f36a}',
537 '\u{1f382}',
538 '\u{1f370}',
539 '\u{1f9c1}',
540 '\u{1f967}',
541 '\u{1f36b}',
542 '\u{1f36c}',
543 '\u{1f36d}',
544 '\u{1f36e}',
545 '\u{1f36f}',
546 '\u{1f37c}',
547 '\u{1f95b}',
548 '\u{2615}',
549 '\u{1fad6}',
550 '\u{1f375}',
551 '\u{1f376}',
552 '\u{1f37e}',
553 '\u{1f377}',
554 '\u{1f378}',
555 '\u{26f0}',
556 '\u{1f30b}',
557 '\u{1f5fb}',
558 '\u{1f3d5}',
559 '\u{1f3d6}',
560 '\u{1f3dc}',
561 '\u{1f3dd}',
562 '\u{1f3de}',
563 '\u{1f3df}',
564 '\u{1f3db}',
565 '\u{1f3d7}',
566 '\u{1f9f1}',
567 '\u{1faa8}',
568 '\u{1fab5}',
569 '\u{1f6d6}',
570 '\u{1f3d8}',
571 '\u{1f3da}',
572 '\u{1f3e0}',
573 '\u{1f3e1}',
574 '\u{1f3e2}',
575 '\u{1f3e3}',
576 '\u{1f3e4}',
577 '\u{1f3e5}',
578 '\u{1f3e6}',
579 '\u{1f3e8}',
580 '\u{1f3e9}',
581 '\u{1f3ea}',
582 '\u{1f3eb}',
583 '\u{1f3ec}',
584 '\u{1f3ed}',
585 '\u{1f3ef}',
586 '\u{1f3f0}',
587 '\u{1f492}',
588 '\u{1f5fc}',
589 '\u{1f5fd}',
590 '\u{26ea}',
591 '\u{1f54c}',
592 '\u{1f6d5}',
593 '\u{1f54d}',
594 '\u{26e9}',
595 '\u{1f54b}',
596 '\u{26f2}',
597 '\u{26fa}',
598 '\u{1f301}',
599 '\u{1f303}',
600 '\u{1f3d9}',
601 '\u{1f304}',
602 '\u{1f305}',
603 '\u{1f306}',
604 '\u{1f307}',
605 '\u{1f309}',
606 '\u{2668}',
607 '\u{1f3a0}',
608 '\u{1f6dd}',
609 '\u{1f3a1}',
610 '\u{1f3a2}',
611 '\u{1f488}',
612 '\u{1f3aa}',
613 '\u{1f682}',
614 '\u{1f326}',
615 '\u{1f327}',
616 '\u{1f328}',
617 '\u{1f329}',
618 '\u{1f32a}',
619 '\u{1f32b}',
620 '\u{1f32c}',
621 '\u{1f300}',
622 '\u{1f308}',
623 '\u{1f302}',
624 '\u{2602}',
625 '\u{2614}',
626 '\u{26f1}',
627 '\u{26a1}',
628 '\u{2744}',
629 '\u{2603}',
630 '\u{26c4}',
631 '\u{2604}',
632 '\u{1f525}',
633 '\u{1f4a7}',
634 '\u{1f30a}',
635 '\u{1f383}',
636 '\u{1f384}',
637 '\u{1f386}',
638 '\u{1f387}',
639 '\u{1f9e8}',
640 '\u{2728}',
641 '\u{1f388}',
642 '\u{1f389}',
643 '\u{1f38a}',
644 '\u{1f38b}',
645 '\u{1f38d}',
646 '\u{1f38e}',
647 '\u{1f38f}',
648 '\u{1f390}',
649 '\u{1f391}',
650 '\u{1f9e7}',
651 '\u{1f380}',
652 '\u{1f381}',
653 '\u{1f397}',
654 '\u{1f39f}',
655 '\u{1f3ab}',
656 '\u{1f396}',
657 '\u{1f3c6}',
658 '\u{1f3c5}',
659 '\u{1f947}',
660 '\u{1f948}',
661 '\u{1f949}',
662 '\u{26bd}',
663 '\u{26be}',
664 '\u{1f94e}',
665 '\u{1f3c0}',
666 '\u{1f3d0}',
667 '\u{1f3c8}',
668 '\u{1f3c9}',
669 '\u{1f3be}',
670 '\u{1f94f}',
671 '\u{1f3b3}',
672 '\u{1f3cf}',
673 '\u{1f3d1}',
674 '\u{1f3d2}',
675 '\u{1f94d}',
676 '\u{1f3d3}',
677 '\u{1f3f8}',
678 '\u{1f94a}',
679 '\u{1f94b}',
680 '\u{1f945}',
681 '\u{26f3}',
682 '\u{26f8}',
683 '\u{1f3a3}',
684 '\u{1f93f}',
685 '\u{1f3bd}',
686 '\u{1f3bf}',
687 '\u{1f6f7}',
688 '\u{1f94c}',
689 '\u{1f3af}',
690 '\u{1fa80}',
691 '\u{1fa81}',
692 '\u{1f52b}',
693 '\u{1f3b1}',
694 '\u{1f52e}',
695 '\u{1fa84}',
696 '\u{1f3ae}',
697 '\u{1f579}',
698 '\u{1f3b0}',
699 '\u{1f3b2}',
700 '\u{1f9e9}',
701 '\u{1f9f8}',
702 '\u{1fa85}',
703 '\u{1faa9}',
704 '\u{1fa86}',
705 '\u{2660}',
706 '\u{2665}',
707 '\u{2666}',
708 '\u{2663}',
709 '\u{265f}',
710 '\u{1f0cf}',
711 '\u{1f004}',
712 '\u{1f3b4}',
713 '\u{1f3ad}',
714 '\u{1f5bc}',
715 '\u{1f3a8}',
716 '\u{1f9f5}',
717 '\u{1faa1}',
718 '\u{1f9f6}',
719 '\u{1faa2}',
720 '\u{1f453}',
721 '\u{1f576}',
722 '\u{1f97d}',
723 '\u{1f97c}',
724 '\u{1f9ba}',
725 '\u{1f454}',
726 '\u{1f455}',
727 '\u{1f456}',
728 '\u{1f9e3}',
729 '\u{1f9e4}',
730 '\u{1f9e5}',
731 '\u{1f9e6}',
732 '\u{1f457}',
733 '\u{1f458}',
734 '\u{1f97b}',
735 '\u{1fa71}',
736 '\u{1fa72}',
737 '\u{1fa73}',
738 '\u{1f459}',
739 '\u{1f45a}',
740 '\u{1faad}',
741 '\u{1f45b}',
742 '\u{1f45c}',
743 '\u{1f45d}',
744 '\u{1f6cd}',
745 '\u{1f392}',
746 '\u{1fa74}',
747 '\u{1f45e}',
748 '\u{1f45f}',
749 '\u{1f97e}',
750 '\u{1f97f}',
751 '\u{1f460}',
752 '\u{1f461}',
753 '\u{1fa70}',
754 '\u{1f462}',
755 '\u{1faae}',
756 '\u{1f451}',
757 '\u{1f452}',
758 '\u{1f3a9}',
759 '\u{1f393}',
760 '\u{1f9e2}',
761 '\u{1fa96}',
762 '\u{26d1}',
763 '\u{1f4ff}',
764 '\u{1f484}',
765 '\u{1f48d}',
766 '\u{1f48e}',
767 '\u{1f507}',
768 '\u{1f508}',
769 '\u{1f509}',
770 '\u{1f50a}',
771 '\u{1f4e2}',
772 '\u{1f4e3}',
773 '\u{1f4ef}',
774 '\u{1f514}',
775 '\u{1f515}',
776 '\u{1f3bc}',
777 '\u{1f3b5}',
778 '\u{1f3b6}',
779 '\u{1f399}',
780 '\u{1f39a}',
781 '\u{1f39b}',
782 '\u{1f3a4}',
783 '\u{1f3a7}',
784 '\u{1f4fb}',
785 '\u{1f3b7}',
786 '\u{1fa97}',
787 '\u{1f3b8}',
788 '\u{1f3b9}',
789 '\u{1f3ba}',
790 '\u{1f3bb}',
791 '\u{1fa95}',
792 '\u{1f941}',
793 '\u{1fa98}',
794 '\u{1fa87}',
795 '\u{1fa88}',
796 '\u{1f4f1}',
797 '\u{1f4f2}',
798 '\u{260e}',
799 '\u{1f4de}',
800 '\u{1f4df}',
801 '\u{1f4e0}',
802 '\u{1f50b}',
803 '\u{1faab}',
804 '\u{1f50c}',
805 '\u{1f4bb}',
806 '\u{1f5a5}',
807 '\u{1f5a8}',
808 '\u{2328}',
809 '\u{1f5b1}',
810 '\u{1f5b2}',
811 '\u{1f4bd}',
812 '\u{1f4be}',
813 '\u{1f4bf}',
814 '\u{1f4c0}',
815 '\u{1f9ee}',
816 '\u{1f3a5}',
817 '\u{1f39e}',
818 '\u{1f4fd}',
819 '\u{1f3ac}',
820 '\u{1f4fa}',
821 '\u{1f4f7}',
822 '\u{1f4f8}',
823 '\u{1f4f9}',
824 '\u{1f4fc}',
825 '\u{1f50d}',
826 '\u{1f50e}',
827 '\u{1f56f}',
828 '\u{1f4a1}',
829 '\u{1f526}',
830 '\u{1f3ee}',
831 '\u{1fa94}',
832 '\u{1f4d4}',
833 '\u{1f4d5}',
834 '\u{1f4d6}',
835 '\u{1f4d7}',
836 '\u{1f4d8}',
837 '\u{1f4d9}',
838 '\u{1f4da}',
839 '\u{1f4d3}',
840 '\u{1f4d2}',
841 '\u{1f4c3}',
842 '\u{1f4dc}',
843 '\u{1f4c4}',
844 '\u{1f4f0}',
845 '\u{1f5de}',
846 '\u{1f4d1}',
847 '\u{1f516}',
848 '\u{1f3f7}',
849 '\u{1f4b0}',
850 '\u{1fa99}',
851 '\u{1f4b4}',
852 '\u{1f4b5}',
853 '\u{1f4b6}',
854 '\u{1f4b7}',
855 '\u{1f4b8}',
856 '\u{1f4b3}',
857 '\u{1f9fe}',
858 '\u{1f4b9}',
859 '\u{2709}',
860 '\u{1f4e7}',
861 '\u{1f4e8}',
862 '\u{1f4e9}',
863 '\u{1f4e4}',
864 '\u{1f4e5}',
865 '\u{1f4e6}',
866 '\u{1f4eb}',
867 '\u{1f4ea}',
868 '\u{1f4ec}',
869 '\u{1f4ed}',
870 '\u{1f4ee}',
871 '\u{1f5f3}',
872 '\u{270f}',
873 ];
874 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
875
876 test!("Here we demonstrate a unicode alphabet of {} emojis with a token size of {}.",
877 A_LEN, X_LEN);
878 test!("Because the token size exceeds 8 bits, this yields some compression.");
879 test!("Unicode only has 1,424 in 2024, so in order to represent, say unique ids with");
880 test!(" just a few emojis, we'll need to create a larger, custom set of glyphs.");
881 let first = 10;
882 test!("Here are the first {} characters of the alphabet:", first);
883 for token in 0..first {
884 let c = base2x.get_char(token);
885 let width = X_LEN;
886 test!(" '{:0width$b}' -> {}", token, c);
887 } // 12345678901234567890123456789012
888 let input = "Every moment is a new beginning.";
889 let input_byts = input.as_bytes();
890 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
891 let mut s = String::new();
892 for byt in input_byts {
893 test!(" {:08b}", byt);
894 s.push_str(&fmt!("{:08b}", byt));
895 }
896 let g = Stringer::new(s);
897 test!("As a bit string: {}", g.insert_every("_", 8));
898 test!("As a bit string: {}", g.insert_every("_", X_LEN));
899 test!("Performing Base2x encoding from bytes to string...");
900 let encoded = base2x.to_string(&input_byts[..]);
901 test!("Base2x encoding: '{}'", encoded);
902 test!("Performing Base2x decoding from string to bytes...");
903 let byts = res!(base2x.from_str(&encoded));
904 let decoded = res!(std::str::from_utf8(&byts));
905 test!("When the bytes are converted back to a string we get: '{}'", decoded);
906 req!(decoded, input);
907 Ok(())
908 }));
909
910 Ok(())
911}