oxedyne/fe2o3/fe2o3_text/tests/regex_oracle/rust_suite.py
4.1 KiB, 1 run
created by r1870400018:60348, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | #!/usr/bin/env python3 |
| 2 | """Converts the Rust `regex` crate's own test suite into rust_suite.txt for tests/regex.rs. |
| 3 | |
| 4 | The crate's `testdata/*.toml` files state what the crate -- and so Typst's `regex(...)` -- answers |
| 5 | for each pattern and haystack. Only the cases that ask what fe2o3_text's engine offers are kept: |
| 6 | leftmost-first searching over UTF-8 text with Unicode on, no bounds, no anchoring, no regex sets |
| 7 | and no byte-oriented `(?-u)` patterns. The data is the crate's, under its MIT or Apache-2.0 |
| 8 | licence. |
| 9 | |
| 10 | Run from this directory, naming one or more of the crate's source directories; a case named in |
| 11 | an earlier one is not taken again from a later one. 1.11.1 still carries the AT&T "fowler" |
| 12 | suites that later releases dropped from the package: |
| 13 | |
| 14 | python3 rust_suite.py ~/.cargo/registry/src/*/regex-1.13.1 \\ |
| 15 | ~/.cargo/registry/src/*/regex-1.11.1 > rust_suite.txt |
| 16 | """ |
| 17 | import pathlib |
| 18 | import re |
| 19 | import sys |
| 20 | import tomllib |
| 21 | |
| 22 | def esc(s): |
| 23 | return s.replace('\\', '\\\\').replace('\t', '\\t').replace('\n', '\\n').replace('\r', '\\r') |
| 24 | |
| 25 | def unescape(s): |
| 26 | # The crate's own unescaping: \xNN bytes, \n, \t and friends. Bytes that are not UTF-8 make |
| 27 | # the case one for byte haystacks, which the caller drops. |
| 28 | out = bytearray() |
| 29 | i = 0 |
| 30 | b = s.encode('utf-8') |
| 31 | while i < len(b): |
| 32 | c = b[i:i+1] |
| 33 | if c == b'\\' and i + 1 < len(b): |
| 34 | n = b[i+1:i+2] |
| 35 | if n == b'x' and i + 3 < len(b) + 1: |
| 36 | out += bytes([int(b[i+2:i+4], 16)]) |
| 37 | i += 4 |
| 38 | continue |
| 39 | out += {b'n': b'\n', b't': b'\t', b'r': b'\r', b'\\': b'\\', b'0': b'\0'}.get(n, b'\\' + n) |
| 40 | i += 2 |
| 41 | continue |
| 42 | out += c |
| 43 | i += 1 |
| 44 | return out.decode('utf-8') |
| 45 | |
| 46 | def unsupported_flags(rx): |
| 47 | # `R` (CRLF lines) is not offered, and `-u` turns Unicode off, which this engine never does. |
| 48 | for g in re.findall(r'\(\?([a-zA-Z-]+)[:)]', rx): |
| 49 | on, _, off = g.partition('-') |
| 50 | if 'R' in g or 'u' in off: |
| 51 | return True |
| 52 | return False |
| 53 | |
| 54 | def emit(key, t): |
| 55 | """Writes one case, or returns False when it asks for something the engine does not offer.""" |
| 56 | rx = t.get('regex') |
| 57 | anchored = bool(t.get('anchored')) |
| 58 | if (not isinstance(rx, str) |
| 59 | or t.get('utf8', True) is False |
| 60 | or t.get('unicode', True) is False |
| 61 | or 'bounds' in t or 'line-terminator' in t |
| 62 | or (anchored and len(t.get('matches', [])) > 1) |
| 63 | or t.get('match-kind', 'leftmost-first') != 'leftmost-first' |
| 64 | or t.get('search-kind', 'leftmost') != 'leftmost' |
| 65 | or unsupported_flags(rx)): |
| 66 | return False |
| 67 | hay = t.get('haystack', '') |
| 68 | if t.get('unescape'): |
| 69 | try: |
| 70 | hay = unescape(hay) |
| 71 | except (UnicodeDecodeError, ValueError): |
| 72 | return False |
| 73 | flags = ('i' if t.get('case-insensitive') else '') + ('a' if anchored else '') |
| 74 | print('T\t' + key + '\t' + (flags or '-')) |
| 75 | print('P\t' + esc(rx)) |
| 76 | print('H\t' + esc(hay)) |
| 77 | if t.get('compiles', True) is False: |
| 78 | print('C') |
| 79 | else: |
| 80 | limit = t.get('match-limit') |
| 81 | print('L\t' + (str(limit) if limit is not None else '-')) |
| 82 | for m in t.get('matches', []): |
| 83 | if isinstance(m, dict): |
| 84 | m = m['span'] |
| 85 | if m and isinstance(m[0], list): |
| 86 | print('M\t' + ' '.join(f'{g[0]},{g[1]}' if g else '-' for g in m)) |
| 87 | else: |
| 88 | print('M\t' + f'{m[0]},{m[1]}') |
| 89 | print('E') |
| 90 | return True |
| 91 | |
| 92 | seen = set() |
| 93 | kept = skipped = 0 |
| 94 | for root in sys.argv[1:]: |
| 95 | base = pathlib.Path(root) / 'testdata' |
| 96 | for path in sorted(base.rglob('*.toml')): |
| 97 | rel = str(path.relative_to(base)) |
| 98 | # regex-lite is another crate, ASCII-only by design; its cases are not the regex crate's. |
| 99 | if rel == 'regex-lite.toml': |
| 100 | continue |
| 101 | for t in tomllib.loads(path.read_text()).get('test', []): |
| 102 | key = rel + '/' + t.get('name', '') |
| 103 | if key in seen: |
| 104 | continue |
| 105 | seen.add(key) |
| 106 | if emit(key, t): |
| 107 | kept += 1 |
| 108 | else: |
| 109 | skipped += 1 |
| 110 | print(f'{kept} kept, {skipped} skipped', file=sys.stderr) |