Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/tests/base2x.rs

28.8 KiB, 14 runs

created by r1870400018:1115, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1use oxedyne_fe2o3_text::{
2 base2x::{
3 self,
4 Base2x,
5 },
6 string::Stringer,
7};
8
9use oxedyne_fe2o3_core::{
10 prelude::*,
11 test::test_it,
12};
13
14use base64;
15
16
17// Default constants.
18const X_LEN: usize = 3;
19const A_LEN: usize = base2x::alphabet_size(X_LEN as u32);
20
21pub fn test_base2x(filter: &'static str) -> Outcome<()> {
22
23 res!(test_it(filter, &["Simple create 000", "all", "base2x"], || {
24 const A_LEN: usize = 9;
25 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc123"));
26 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
27 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
28 Ok(_base2x) => return Err(err!(
29 "Alphabet of length {} should be invalid.", alphabet.len();
30 Invalid, Input, Test)),
31 Err(e) => msg!("Correctly triggered: {}", e),
32 };
33 Ok(())
34 }));
35
36 res!(test_it(filter, &["Simple create 010", "all", "base2x"], || {
37 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc12"));
38 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
39 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
40 Ok(_base2x) => (),
41 Err(e) => return Err(e),
42 };
43 Ok(())
44 }));
45
46 res!(test_it(filter, &["Simple create 020", "all", "base2x"], || {
47 const A_LEN: usize = 300;
48 let repeated_char = 'A';
49 let repeat_count = 300;
50 let alphabet: String =
51 std::iter::repeat(repeated_char).take(repeat_count).collect();
52
53 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet(&alphabet));
54 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
55 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
56 Ok(_base2x) => return Err(err!(
57 "Alphabet of length {} should be invalid.", alphabet.len();
58 Invalid, Input, Test)),
59 Err(e) => test!("Correctly triggered: {}", e),
60 };
61 Ok(())
62 }));
63
64 res!(test_it(filter, &["Simple create 030", "all", "base2x", "unicode"], || {
65 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCab\u{e9}12"));
66 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("123"));
67 match Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))) {
68 Ok(_base2x) => (),
69 Err(e) => return Err(e),
70 };
71 Ok(())
72 }));
73
74 res!(test_it(filter, &["Round trip 000", "all", "base2x", "hematite"], || {
75 let base2x = base2x::HEMATITE64;
76 let a = base2x.alphabet_size();
77 let x = base2x.token_size();
78 let input = fmt!("This Hematite alphabet has {} characters", a);
79 let input_byts = input.as_bytes();
80 trace!("This is the alphabet we'll be using:");
81 for line in base2x.fmt_char_map() {
82 trace!(" {}", line);
83 }
84 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
85 for byt in input_byts {
86 debug!(" {:08b}", byt);
87 }
88 test!("Each of the {} tokens has a length of {} bits.", a, x);
89 test!("Performing Base2x encoding from bytes to string...");
90 let encoded = base2x.to_string(&input_byts);
91 test!("Base2x encoding: '{}'", encoded);
92
93 test!("Performing Base2x decoding from string to bytes...");
94 let byts = res!(base2x.from_str(&encoded));
95 let mut s = String::new();
96 debug!("This gives us the following bytes:");
97 for byt in &byts {
98 debug!(" {:08b}", byt);
99 s.push_str(&fmt!("{:08b}", byt));
100 }
101 let g = Stringer::new(s);
102 debug!("As a bit string: {}", g.insert_every("_", 8));
103 debug!("As a bit string: {}", g.insert_every("_", x));
104
105 let decoded = res!(std::str::from_utf8(&byts));
106 test!("When the bytes are converted back to a string we get: '{}'", decoded);
107 req!(decoded, input);
108 Ok(())
109 }));
110
111 res!(test_it(filter, &["Round trip 005", "all", "base2x", "hematite"], || {
112 let base2x = base2x::HEMATITE32;
113 let a = base2x.alphabet_size();
114 let x = base2x.token_size();
115 let input = fmt!("This Hematite alphabet has {} characters", a);
116 let input_byts = input.as_bytes();
117 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
118 for byt in input_byts {
119 debug!(" {:08b}", byt);
120 }
121 test!("Each of the {} tokens has a length of {} bits.", a, x);
122 test!("Performing Base2x encoding from bytes to string...");
123 let encoded = base2x.to_string(&input_byts);
124 test!("Base2x encoding: '{}'", encoded);
125
126 test!("Performing Base2x decoding from string to bytes...");
127 debug!("This gives us the following bytes:");
128 let byts = res!(base2x.from_str(&encoded));
129 let mut s = String::new();
130 for byt in &byts {
131 debug!(" {:08b}", byt);
132 s.push_str(&fmt!("{:08b}", byt));
133 }
134 let g = Stringer::new(s);
135 debug!("As a bit string: {}", g.insert_every("_", 8));
136 debug!("As a bit string: {}", g.insert_every("_", x));
137
138 let decoded = res!(std::str::from_utf8(&byts));
139 test!("When the bytes are converted back to a string we get: '{}'", decoded);
140 req!(decoded, input);
141 Ok(())
142 }));
143
144 res!(test_it(filter, &["Round trip 010", "all", "base2x"], || {
145 let base2x = base2x::BASE64;
146 let a = base2x.alphabet_size();
147 let x = base2x.token_size();
148 for line in base2x.fmt_char_map() {
149 debug!("{}", line);
150 }
151
152 let input = "Man";
153 test!("This is an example from the Wikipedia page for Base64.");
154 let input_byts = input.as_bytes();
155 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
156 for byt in input_byts {
157 test!(" {:08b}", byt);
158 }
159 test!("Each of the {} tokens has a length of {} bits.", a, x);
160 test!("Performing Base2x encoding from bytes to string...");
161 let encoded = base2x.to_string(&input_byts);
162 test!("Base2x encoding: '{}'", encoded);
163
164 let base64_encoded = base64::encode(&input_byts);
165 test!("Now let's encode using the standard alphabet of Base64: '{}'", base64_encoded);
166 test!("They should be identical, because there is no padding in this case.");
167 req!(encoded, base64_encoded);
168 test!("The character token mapping is:");
169 for c in encoded.chars() {
170 let token_opt = base2x.get_token(c);
171 test!(" '{}' -> {}", c,
172 token_opt.map_or(
173 fmt!("None"),
174 |b| fmt!("{:0width$b}", b, width = x),
175 ),
176 );
177 }
178
179 test!("Performing Base2x decoding from string to bytes...");
180 let byts = res!(base2x.from_str(&encoded));
181 test!("This gives us the following bytes:");
182 let mut s = String::new();
183 for byt in &byts {
184 test!(" {:08b}", byt);
185 s.push_str(&fmt!("{:08b}", byt));
186 }
187 let g = Stringer::new(s);
188 test!("As a bit string: {}", g.insert_every("_", 8));
189 test!("As a bit string: {}", g.insert_every("_", x));
190
191 test!("Performing Base2x encoding back from bytes to string...");
192 let decoded = res!(std::str::from_utf8(&byts));
193 test!("When the bytes are converted back to a string we get: '{}'", decoded);
194 req!(decoded, input);
195 Ok(())
196 }));
197
198 res!(test_it(filter, &["Round trip 011", "all", "base2x"], || {
199 let base2x = base2x::BASE64;
200 let a = base2x.alphabet_size();
201 let x = base2x.token_size();
202 for line in base2x.fmt_char_map() {
203 debug!("{}", line);
204 }
205
206 let input = "Ma";
207 test!("This is an example from the Wikipedia page for Base64.");
208 let input_byts = input.as_bytes();
209 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
210 for byt in input_byts {
211 test!(" {:08b}", byt);
212 }
213 test!("Each of the {} tokens has a length of {} bits.", a, x);
214 test!("Performing Base2x encoding from bytes to string...");
215 let encoded = base2x.to_string(&input_byts);
216 test!("Base2x encoding: '{}'", encoded);
217 let mut compare = encoded.clone();
218 compare.pop();
219 test!("This should include padding, so let's remove the Base2x-specific part: '{}'", compare);
220 let base64_encoded = base64::encode(&input_byts);
221 test!("The character token mapping is:");
222 for c in encoded.chars() {
223 let token_opt = base2x.get_token(c);
224 test!(" '{}' -> {}", c,
225 token_opt.map_or(
226 fmt!("None"),
227 |b| fmt!("{:0width$b}", b, width = x),
228 ),
229 );
230 }
231 test!("Now let's encode using the standard alphabet of Base64: '{}'", base64_encoded);
232 test!("They should be identical, because they use a similar padding scheme.");
233 req!(compare, base64_encoded);
234
235 test!("Performing Base2x decoding from string to bytes...");
236 let byts = res!(base2x.from_str(&encoded));
237 test!("This gives us the following bytes:");
238 let mut s = String::new();
239 for byt in &byts {
240 test!(" {:08b}", byt);
241 s.push_str(&fmt!("{:08b}", byt));
242 }
243 let g = Stringer::new(s);
244 test!("As a bit string: {}", g.insert_every("_", 8));
245 test!("As a bit string: {}", g.insert_every("_", x));
246 test!("You can see at the end here that the last character is partial, with only 4 bits.");
247 test!("The padding appends 2 zero bits, giving 000100 or 'E'.");
248
249 test!("Performing Base2x encoding back from bytes to string...");
250 let decoded = res!(std::str::from_utf8(&byts));
251 test!("When the bytes are converted back to a string we get: '{}'", decoded);
252 req!(decoded, input);
253 Ok(())
254 }));
255
256 res!(test_it(filter, &["Round trip 100", "all", "base2x"], || {
257 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet("ABCabc12"));
258 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
259 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
260 test!("This is the alphabet we'll be using:");
261 for line in base2x.fmt_char_map() {
262 test!(" {}", line);
263 }
264 let mut input = "BA2c1baBCaa21b111AcCb".to_string();
265 test!("Let's start with an arbitrary Base2x encoding: '{}'", input);
266 test!("Since each token is {} bits long, that's {} bits which is not divisible by 8.",
267 X_LEN, input.len() * X_LEN);
268 test!("Which means padding is needed, so let's apply it using Base2x normalisation...");
269 let padding = base2x.normalise(&mut input);
270 if padding > 0 {
271 base2x.push_pad(&mut input, padding);
272 }
273 test!("Normalised: '{}' with {} bits of padding.", input, padding);
274 test!("The token values for each encoded character are:");
275 for c in input.chars() {
276 let token_opt = base2x.get_token(c);
277 test!(" '{}' -> {}", c,
278 token_opt.map_or(
279 fmt!("None"),
280 |b| fmt!("{:0width$b}", b, width = X_LEN),
281 ),
282 );
283 }
284 test!("Performing Base2x decoding from string to bytes...");
285 let byts = res!(base2x.from_str(&input));
286 test!("This gives us the following bytes:");
287 let mut s = String::new();
288 for byt in &byts {
289 test!(" {:08b}", byt);
290 s.push_str(&fmt!("{:08b}", byt));
291 }
292 let g = Stringer::new(s);
293 test!("As a bit string: {}", g.insert_every("_", 8));
294 test!("As a bit string: {}", g.insert_every("_", X_LEN));
295
296 test!("Performing Base2x encoding back from bytes to string...");
297 let encoded = base2x.to_string(&byts);
298 test!("When the bytes are Base2x encoded back to a string we get: '{}'", encoded);
299 req!(encoded, input);
300 Ok(())
301 }));
302
303 res!(test_it(filter, &["Round trip 200", "all", "base2x"], || {
304 let byts = vec![0b01001011, 0b11101000, 0b10110100, 0b01101111];
305 test!("input bytes:");
306 for byt in &byts {
307 test!("{:08b}", byt);
308 }
309 let alphabet = "ABCabc12";
310 test!("alphabet = '{}'", alphabet);
311 let alphabet = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_alphabet(alphabet));
312 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12_"));
313 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
314 let n = byts.len();
315 let n_bits = 8*n;
316 let n_tokens = n_bits / base2x.token_size();
317 let rem = n_bits - n_tokens * base2x.token_size();
318 test!("{} input bytes have {} bits, requring {} tokens with a remainder of {}",
319 n, n_bits, n_tokens, rem);
320 let mut s = String::new();
321 for byt in &byts {
322 s.push_str(&fmt!("{:08b}", byt));
323 }
324 let g = Stringer::new(s);
325 test!("bit string: {}", g.insert_every("_", 8));
326 test!("bit string: {}", g.insert_every("_", 3));
327 let encoded = base2x.to_string(&byts);
328 test!("encoded = '{}'", encoded);
329
330 test!("encoded '{}' has {} characters, requring {} bits",
331 encoded, encoded.len(), base2x.token_size() * encoded.len());
332 for c in encoded.chars() {
333 let token_opt = base2x.get_token(c);
334 test!(" '{}' -> {}", c,
335 token_opt.map_or(
336 fmt!("None"),
337 |b| fmt!("{:0width$b}", b, width = base2x.token_size()),
338 ),
339 );
340 }
341 // Now decode string -> bytes
342 let byts2 = res!(base2x.from_str(&encoded));
343 let mut s = String::new();
344 for byt in &byts2 {
345 s.push_str(&fmt!("{:08b}", byt));
346 }
347 let g = Stringer::new(s);
348 test!("decoded bit string: {}", g.insert_every("_", 8));
349 test!("decoded bit string: {}", g.insert_every("_", base2x.token_size()));
350 req!(byts2, byts);
351 Ok(())
352 }));
353
354 res!(test_it(filter, &["Round trip 300", "all", "base2x", "unicode"], || {
355 const X_LEN: usize = 9;
356 const A_LEN: usize = base2x::alphabet_size(X_LEN as u32);
357 let pad_set = res!(Base2x::<{A_LEN}, {X_LEN}>::prepare_pad_set("12345678_"));
358 let alphabet = [
359 '\u{1f600}',
360 '\u{1f976}',
361 '\u{1f4a9}',
362 '\u{1f63b}',
363 '\u{1f44c}',
364 '\u{1f441}',
365 '\u{1f937}',
366 '\u{1f3cb}',
367 '\u{1f984}',
368 '\u{1f30f}',
369 '\u{26f5}',
370 '\u{2708}',
371 '\u{1fa90}',
372 '\u{1fabf}',
373 '\u{1f438}',
374 '\u{1f40a}',
375 '\u{1f422}',
376 '\u{1f98e}',
377 '\u{1f40d}',
378 '\u{1f432}',
379 '\u{1f409}',
380 '\u{1f995}',
381 '\u{1f996}',
382 '\u{1f433}',
383 '\u{1f40b}',
384 '\u{1f42c}',
385 '\u{1f9ad}',
386 '\u{1f41f}',
387 '\u{1f420}',
388 '\u{1f421}',
389 '\u{1f988}',
390 '\u{1f419}',
391 '\u{1f41a}',
392 '\u{1fab8}',
393 '\u{1fabc}',
394 '\u{1f40c}',
395 '\u{1f98b}',
396 '\u{1f41b}',
397 '\u{1f41c}',
398 '\u{1f41d}',
399 '\u{1fab2}',
400 '\u{1f41e}',
401 '\u{1f997}',
402 '\u{1fab3}',
403 '\u{1f577}',
404 '\u{1f578}',
405 '\u{1f982}',
406 '\u{1f99f}',
407 '\u{1fab0}',
408 '\u{1fab1}',
409 '\u{1f9a0}',
410 '\u{1f490}',
411 '\u{1f338}',
412 '\u{1f4ae}',
413 '\u{1fab7}',
414 '\u{1f3f5}',
415 '\u{1f339}',
416 '\u{1f940}',
417 '\u{1f33a}',
418 '\u{1f33b}',
419 '\u{1f33c}',
420 '\u{1f337}',
421 '\u{1fabb}',
422 '\u{1f331}',
423 '\u{1fab4}',
424 '\u{1f332}',
425 '\u{1f333}',
426 '\u{1f334}',
427 '\u{1f335}',
428 '\u{1f33e}',
429 '\u{1f33f}',
430 '\u{2618}',
431 '\u{1f340}',
432 '\u{1f341}',
433 '\u{1f342}',
434 '\u{1f343}',
435 '\u{1fab9}',
436 '\u{1faba}',
437 '\u{1f344}',
438 '\u{1f347}',
439 '\u{1f348}',
440 '\u{1f349}',
441 '\u{1f34a}',
442 '\u{1f34b}',
443 '\u{1f34c}',
444 '\u{1f34d}',
445 '\u{1f96d}',
446 '\u{1f34e}',
447 '\u{1f34f}',
448 '\u{1f350}',
449 '\u{1f351}',
450 '\u{1f352}',
451 '\u{1f353}',
452 '\u{1fad0}',
453 '\u{1f95d}',
454 '\u{1f345}',
455 '\u{1fad2}',
456 '\u{1f965}',
457 '\u{1f951}',
458 '\u{1f346}',
459 '\u{1f954}',
460 '\u{1f955}',
461 '\u{1f33d}',
462 '\u{1f336}',
463 '\u{1fad1}',
464 '\u{1f952}',
465 '\u{1f96c}',
466 '\u{1f966}',
467 '\u{1f9c4}',
468 '\u{1f9c5}',
469 '\u{1f95c}',
470 '\u{1fad8}',
471 '\u{1f330}',
472 '\u{1fada}',
473 '\u{1fadb}',
474 '\u{1f35e}',
475 '\u{1f950}',
476 '\u{1f956}',
477 '\u{1fad3}',
478 '\u{1f968}',
479 '\u{1f96f}',
480 '\u{1f95e}',
481 '\u{1f9c7}',
482 '\u{1f9c0}',
483 '\u{1f356}',
484 '\u{1f357}',
485 '\u{1f969}',
486 '\u{1f953}',
487 '\u{1f354}',
488 '\u{1f35f}',
489 '\u{1f355}',
490 '\u{1f32d}',
491 '\u{1f96a}',
492 '\u{1f32e}',
493 '\u{1f32f}',
494 '\u{1fad4}',
495 '\u{1f959}',
496 '\u{1f9c6}',
497 '\u{1f95a}',
498 '\u{1f373}',
499 '\u{1f958}',
500 '\u{1f372}',
501 '\u{1fad5}',
502 '\u{1f963}',
503 '\u{1f957}',
504 '\u{1f37f}',
505 '\u{1f9c8}',
506 '\u{1f9c2}',
507 '\u{1f96b}',
508 '\u{1f371}',
509 '\u{1f358}',
510 '\u{1f359}',
511 '\u{1f35a}',
512 '\u{1f35b}',
513 '\u{1f35c}',
514 '\u{1f35d}',
515 '\u{1f360}',
516 '\u{1f362}',
517 '\u{1f363}',
518 '\u{1f364}',
519 '\u{1f365}',
520 '\u{1f96e}',
521 '\u{1f361}',
522 '\u{1f95f}',
523 '\u{1f960}',
524 '\u{1f961}',
525 '\u{1f980}',
526 '\u{1f99e}',
527 '\u{1f990}',
528 '\u{1f991}',
529 '\u{1f9aa}',
530 '\u{1f366}',
531 '\u{1f367}',
532 '\u{1f368}',
533 '\u{1f369}',
534 '\u{1f36a}',
535 '\u{1f382}',
536 '\u{1f370}',
537 '\u{1f9c1}',
538 '\u{1f967}',
539 '\u{1f36b}',
540 '\u{1f36c}',
541 '\u{1f36d}',
542 '\u{1f36e}',
543 '\u{1f36f}',
544 '\u{1f37c}',
545 '\u{1f95b}',
546 '\u{2615}',
547 '\u{1fad6}',
548 '\u{1f375}',
549 '\u{1f376}',
550 '\u{1f37e}',
551 '\u{1f377}',
552 '\u{1f378}',
553 '\u{26f0}',
554 '\u{1f30b}',
555 '\u{1f5fb}',
556 '\u{1f3d5}',
557 '\u{1f3d6}',
558 '\u{1f3dc}',
559 '\u{1f3dd}',
560 '\u{1f3de}',
561 '\u{1f3df}',
562 '\u{1f3db}',
563 '\u{1f3d7}',
564 '\u{1f9f1}',
565 '\u{1faa8}',
566 '\u{1fab5}',
567 '\u{1f6d6}',
568 '\u{1f3d8}',
569 '\u{1f3da}',
570 '\u{1f3e0}',
571 '\u{1f3e1}',
572 '\u{1f3e2}',
573 '\u{1f3e3}',
574 '\u{1f3e4}',
575 '\u{1f3e5}',
576 '\u{1f3e6}',
577 '\u{1f3e8}',
578 '\u{1f3e9}',
579 '\u{1f3ea}',
580 '\u{1f3eb}',
581 '\u{1f3ec}',
582 '\u{1f3ed}',
583 '\u{1f3ef}',
584 '\u{1f3f0}',
585 '\u{1f492}',
586 '\u{1f5fc}',
587 '\u{1f5fd}',
588 '\u{26ea}',
589 '\u{1f54c}',
590 '\u{1f6d5}',
591 '\u{1f54d}',
592 '\u{26e9}',
593 '\u{1f54b}',
594 '\u{26f2}',
595 '\u{26fa}',
596 '\u{1f301}',
597 '\u{1f303}',
598 '\u{1f3d9}',
599 '\u{1f304}',
600 '\u{1f305}',
601 '\u{1f306}',
602 '\u{1f307}',
603 '\u{1f309}',
604 '\u{2668}',
605 '\u{1f3a0}',
606 '\u{1f6dd}',
607 '\u{1f3a1}',
608 '\u{1f3a2}',
609 '\u{1f488}',
610 '\u{1f3aa}',
611 '\u{1f682}',
612 '\u{1f326}',
613 '\u{1f327}',
614 '\u{1f328}',
615 '\u{1f329}',
616 '\u{1f32a}',
617 '\u{1f32b}',
618 '\u{1f32c}',
619 '\u{1f300}',
620 '\u{1f308}',
621 '\u{1f302}',
622 '\u{2602}',
623 '\u{2614}',
624 '\u{26f1}',
625 '\u{26a1}',
626 '\u{2744}',
627 '\u{2603}',
628 '\u{26c4}',
629 '\u{2604}',
630 '\u{1f525}',
631 '\u{1f4a7}',
632 '\u{1f30a}',
633 '\u{1f383}',
634 '\u{1f384}',
635 '\u{1f386}',
636 '\u{1f387}',
637 '\u{1f9e8}',
638 '\u{2728}',
639 '\u{1f388}',
640 '\u{1f389}',
641 '\u{1f38a}',
642 '\u{1f38b}',
643 '\u{1f38d}',
644 '\u{1f38e}',
645 '\u{1f38f}',
646 '\u{1f390}',
647 '\u{1f391}',
648 '\u{1f9e7}',
649 '\u{1f380}',
650 '\u{1f381}',
651 '\u{1f397}',
652 '\u{1f39f}',
653 '\u{1f3ab}',
654 '\u{1f396}',
655 '\u{1f3c6}',
656 '\u{1f3c5}',
657 '\u{1f947}',
658 '\u{1f948}',
659 '\u{1f949}',
660 '\u{26bd}',
661 '\u{26be}',
662 '\u{1f94e}',
663 '\u{1f3c0}',
664 '\u{1f3d0}',
665 '\u{1f3c8}',
666 '\u{1f3c9}',
667 '\u{1f3be}',
668 '\u{1f94f}',
669 '\u{1f3b3}',
670 '\u{1f3cf}',
671 '\u{1f3d1}',
672 '\u{1f3d2}',
673 '\u{1f94d}',
674 '\u{1f3d3}',
675 '\u{1f3f8}',
676 '\u{1f94a}',
677 '\u{1f94b}',
678 '\u{1f945}',
679 '\u{26f3}',
680 '\u{26f8}',
681 '\u{1f3a3}',
682 '\u{1f93f}',
683 '\u{1f3bd}',
684 '\u{1f3bf}',
685 '\u{1f6f7}',
686 '\u{1f94c}',
687 '\u{1f3af}',
688 '\u{1fa80}',
689 '\u{1fa81}',
690 '\u{1f52b}',
691 '\u{1f3b1}',
692 '\u{1f52e}',
693 '\u{1fa84}',
694 '\u{1f3ae}',
695 '\u{1f579}',
696 '\u{1f3b0}',
697 '\u{1f3b2}',
698 '\u{1f9e9}',
699 '\u{1f9f8}',
700 '\u{1fa85}',
701 '\u{1faa9}',
702 '\u{1fa86}',
703 '\u{2660}',
704 '\u{2665}',
705 '\u{2666}',
706 '\u{2663}',
707 '\u{265f}',
708 '\u{1f0cf}',
709 '\u{1f004}',
710 '\u{1f3b4}',
711 '\u{1f3ad}',
712 '\u{1f5bc}',
713 '\u{1f3a8}',
714 '\u{1f9f5}',
715 '\u{1faa1}',
716 '\u{1f9f6}',
717 '\u{1faa2}',
718 '\u{1f453}',
719 '\u{1f576}',
720 '\u{1f97d}',
721 '\u{1f97c}',
722 '\u{1f9ba}',
723 '\u{1f454}',
724 '\u{1f455}',
725 '\u{1f456}',
726 '\u{1f9e3}',
727 '\u{1f9e4}',
728 '\u{1f9e5}',
729 '\u{1f9e6}',
730 '\u{1f457}',
731 '\u{1f458}',
732 '\u{1f97b}',
733 '\u{1fa71}',
734 '\u{1fa72}',
735 '\u{1fa73}',
736 '\u{1f459}',
737 '\u{1f45a}',
738 '\u{1faad}',
739 '\u{1f45b}',
740 '\u{1f45c}',
741 '\u{1f45d}',
742 '\u{1f6cd}',
743 '\u{1f392}',
744 '\u{1fa74}',
745 '\u{1f45e}',
746 '\u{1f45f}',
747 '\u{1f97e}',
748 '\u{1f97f}',
749 '\u{1f460}',
750 '\u{1f461}',
751 '\u{1fa70}',
752 '\u{1f462}',
753 '\u{1faae}',
754 '\u{1f451}',
755 '\u{1f452}',
756 '\u{1f3a9}',
757 '\u{1f393}',
758 '\u{1f9e2}',
759 '\u{1fa96}',
760 '\u{26d1}',
761 '\u{1f4ff}',
762 '\u{1f484}',
763 '\u{1f48d}',
764 '\u{1f48e}',
765 '\u{1f507}',
766 '\u{1f508}',
767 '\u{1f509}',
768 '\u{1f50a}',
769 '\u{1f4e2}',
770 '\u{1f4e3}',
771 '\u{1f4ef}',
772 '\u{1f514}',
773 '\u{1f515}',
774 '\u{1f3bc}',
775 '\u{1f3b5}',
776 '\u{1f3b6}',
777 '\u{1f399}',
778 '\u{1f39a}',
779 '\u{1f39b}',
780 '\u{1f3a4}',
781 '\u{1f3a7}',
782 '\u{1f4fb}',
783 '\u{1f3b7}',
784 '\u{1fa97}',
785 '\u{1f3b8}',
786 '\u{1f3b9}',
787 '\u{1f3ba}',
788 '\u{1f3bb}',
789 '\u{1fa95}',
790 '\u{1f941}',
791 '\u{1fa98}',
792 '\u{1fa87}',
793 '\u{1fa88}',
794 '\u{1f4f1}',
795 '\u{1f4f2}',
796 '\u{260e}',
797 '\u{1f4de}',
798 '\u{1f4df}',
799 '\u{1f4e0}',
800 '\u{1f50b}',
801 '\u{1faab}',
802 '\u{1f50c}',
803 '\u{1f4bb}',
804 '\u{1f5a5}',
805 '\u{1f5a8}',
806 '\u{2328}',
807 '\u{1f5b1}',
808 '\u{1f5b2}',
809 '\u{1f4bd}',
810 '\u{1f4be}',
811 '\u{1f4bf}',
812 '\u{1f4c0}',
813 '\u{1f9ee}',
814 '\u{1f3a5}',
815 '\u{1f39e}',
816 '\u{1f4fd}',
817 '\u{1f3ac}',
818 '\u{1f4fa}',
819 '\u{1f4f7}',
820 '\u{1f4f8}',
821 '\u{1f4f9}',
822 '\u{1f4fc}',
823 '\u{1f50d}',
824 '\u{1f50e}',
825 '\u{1f56f}',
826 '\u{1f4a1}',
827 '\u{1f526}',
828 '\u{1f3ee}',
829 '\u{1fa94}',
830 '\u{1f4d4}',
831 '\u{1f4d5}',
832 '\u{1f4d6}',
833 '\u{1f4d7}',
834 '\u{1f4d8}',
835 '\u{1f4d9}',
836 '\u{1f4da}',
837 '\u{1f4d3}',
838 '\u{1f4d2}',
839 '\u{1f4c3}',
840 '\u{1f4dc}',
841 '\u{1f4c4}',
842 '\u{1f4f0}',
843 '\u{1f5de}',
844 '\u{1f4d1}',
845 '\u{1f516}',
846 '\u{1f3f7}',
847 '\u{1f4b0}',
848 '\u{1fa99}',
849 '\u{1f4b4}',
850 '\u{1f4b5}',
851 '\u{1f4b6}',
852 '\u{1f4b7}',
853 '\u{1f4b8}',
854 '\u{1f4b3}',
855 '\u{1f9fe}',
856 '\u{1f4b9}',
857 '\u{2709}',
858 '\u{1f4e7}',
859 '\u{1f4e8}',
860 '\u{1f4e9}',
861 '\u{1f4e4}',
862 '\u{1f4e5}',
863 '\u{1f4e6}',
864 '\u{1f4eb}',
865 '\u{1f4ea}',
866 '\u{1f4ec}',
867 '\u{1f4ed}',
868 '\u{1f4ee}',
869 '\u{1f5f3}',
870 '\u{270f}',
871 ];
872 let base2x = res!(Base2x::<{A_LEN}, {X_LEN}>::new(alphabet, Some(('=', pad_set))));
873
874 test!("Here we demonstrate a unicode alphabet of {} emojis with a token size of {}.",
875 A_LEN, X_LEN);
876 test!("Because the token size exceeds 8 bits, this yields some compression.");
877 test!("Unicode only has 1,424 in 2024, so in order to represent, say unique ids with");
878 test!(" just a few emojis, we'll need to create a larger, custom set of glyphs.");
879 let first = 10;
880 test!("Here are the first {} characters of the alphabet:", first);
881 for token in 0..first {
882 let c = base2x.get_char(token);
883 test!(" '{:03b}' -> {}", token, c);
884 } // 12345678901234567890123456789012
885 let input = "Every moment is a new beginning.";
886 let input_byts = input.as_bytes();
887 test!("Let's start with the {} bytes of this text: '{}'", input_byts.len(), input);
888 let mut s = String::new();
889 for byt in input_byts {
890 test!(" {:08b}", byt);
891 s.push_str(&fmt!("{:08b}", byt));
892 }
893 let g = Stringer::new(s);
894 test!("As a bit string: {}", g.insert_every("_", 8));
895 test!("As a bit string: {}", g.insert_every("_", X_LEN));
896 test!("Performing Base2x encoding from bytes to string...");
897 let encoded = base2x.to_string(&input_byts[..]);
898 test!("Base2x encoding: '{}'", encoded);
899 test!("Performing Base2x decoding from string to bytes...");
900 let byts = res!(base2x.from_str(&encoded));
901 let decoded = res!(std::str::from_utf8(&byts));
902 test!("When the bytes are converted back to a string we get: '{}'", decoded);
903 req!(decoded, input);
904 Ok(())
905 }));
906
907 Ok(())
908}