Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_text/src/fmt/lex.rs

39.4 KiB, 37 runs

created by r1870400018:11628, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Generic lexer driven by language-specific token definitions.
2//!
3//! The lexer transforms source text into a flat sequence of `Token`s,
4//! preserving all whitespace and comments as leading trivia on each
5//! token. This ensures round-trip fidelity: every byte of the
6//! source is accounted for.
7//!
8
9use crate::fmt::cst::{
10 Span,
11 Token,
12 TokenKind,
13 Trivia,
14};
15
16use oxedyne_fe2o3_core::prelude::*;
17
18use std::collections::BTreeSet;
19
20
21/// Language-specific token definitions. These drive the generic
22/// lexer so that adding a new language only requires filling in
23/// this structure.
24#[derive(Clone, Debug)]
25pub struct LangTokens {
26 /// Keywords (e.g. `fn`, `let`, `if`, `struct`).
27 pub keywords: BTreeSet<String>,
28 /// Single-line comment prefix (e.g. `//`).
29 pub line_comment: String,
30 /// Block comment open (e.g. `/*`).
31 pub block_comment_open: String,
32 /// Block comment close (e.g. `*/`).
33 pub block_comment_close: String,
34 /// Doc-comment prefixes (e.g. `///`, `//!`).
35 pub doc_comment_prefixes: Vec<String>,
36 /// Multi-character operators, longest first.
37 pub operators: Vec<String>,
38 /// Single character that opens a string literal (e.g. `"`).
39 pub string_delimiters: Vec<char>,
40 /// Character literal delimiter (e.g. `'`).
41 pub char_delimiter: Option<char>,
42 /// Raw string prefix (e.g. `r#"` in Rust). Empty if none.
43 pub raw_string_prefix: String,
44 /// Literal prefixes that bind to the string or character literal
45 /// after them (e.g. `b`, `br`, `c`, `cr` in Rust), longest first.
46 /// Without these a prefix lexes as a separate identifier and the
47 /// formatter separates it from its literal, which does not compile.
48 pub literal_prefixes: Vec<String>,
49 /// Attribute prefix (e.g. `#[` in Rust). Empty if none.
50 pub attribute_prefix: String,
51 /// Lifetime prefix (e.g. `'` when followed by an ident in Rust).
52 pub lifetime_prefix: Option<char>,
53}
54
55/// Rust token definitions.
56pub fn rust_tokens() -> LangTokens {
57 let keywords: BTreeSet<String> = [
58 "as", "async", "await", "break", "const", "continue", "crate",
59 "dyn", "else", "enum", "extern", "false", "fn", "for", "if",
60 "impl", "in", "let", "loop", "match", "mod", "move", "mut",
61 "pub", "ref", "return", "self", "Self", "static", "struct",
62 "super", "trait", "true", "type", "unsafe", "use", "where",
63 "while", "yield",
64 ].iter().map(|s| s.to_string()).collect();
65
66 let operators = vec![
67 // Three-character.
68 "<<=", ">>=", "..=",
69 // Two-character.
70 "->", "=>", "::", "&&", "||", "==", "!=", "<=", ">=",
71 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
72 "..",
73 ].iter().map(|s| s.to_string()).collect();
74
75 let doc_comment_prefixes = vec![
76 "///".to_string(),
77 "//!".to_string(),
78 ];
79
80 LangTokens {
81 keywords,
82 line_comment: "//".to_string(),
83 block_comment_open: "/*".to_string(),
84 block_comment_close: "*/".to_string(),
85 doc_comment_prefixes,
86 operators,
87 string_delimiters: vec!['"'],
88 char_delimiter: Some('\''),
89 raw_string_prefix: "r".to_string(),
90 // Longest first, so `br` is matched before `b`.
91 literal_prefixes: ["br", "cr", "b", "c"]
92 .iter().map(|s| s.to_string()).collect(),
93 attribute_prefix: "#[".to_string(),
94 lifetime_prefix: Some('\''),
95 }
96}
97
98/// C token definitions.
99pub fn c_tokens() -> LangTokens {
100 let keywords: BTreeSet<String> = [
101 "auto", "break", "case", "char", "const", "continue", "default",
102 "do", "double", "else", "enum", "extern", "float", "for", "goto",
103 "if", "inline", "int", "long", "register", "restrict", "return",
104 "short", "signed", "sizeof", "static", "struct", "switch",
105 "typedef", "union", "unsigned", "void", "volatile", "while",
106 "_Bool", "_Complex", "_Imaginary",
107 ].iter().map(|s| s.to_string()).collect();
108
109 let operators = vec![
110 "<<=", ">>=",
111 "->", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
112 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
113 ].iter().map(|s| s.to_string()).collect();
114
115 LangTokens {
116 keywords,
117 line_comment: "//".to_string(),
118 block_comment_open: "/*".to_string(),
119 block_comment_close: "*/".to_string(),
120 doc_comment_prefixes: Vec::new(),
121 operators,
122 string_delimiters: vec!['"'],
123 char_delimiter: Some('\''),
124 raw_string_prefix: String::new(),
125 literal_prefixes: Vec::new(),
126 attribute_prefix: String::new(),
127 lifetime_prefix: None,
128 }
129}
130
131/// C++ token definitions.
132pub fn cpp_tokens() -> LangTokens {
133 let keywords: BTreeSet<String> = [
134 // C keywords carried forward.
135 "auto", "break", "case", "char", "const", "continue", "default",
136 "do", "double", "else", "enum", "extern", "float", "for", "goto",
137 "if", "inline", "int", "long", "register", "return",
138 "short", "signed", "sizeof", "static", "struct", "switch",
139 "typedef", "union", "unsigned", "void", "volatile", "while",
140 // C++ additions.
141 "alignas", "alignof", "and", "and_eq", "asm", "bitand", "bitor",
142 "bool", "catch", "class", "compl", "concept", "consteval",
143 "constexpr", "constinit", "const_cast", "co_await", "co_return",
144 "co_yield", "decltype", "delete", "dynamic_cast", "explicit",
145 "export", "false", "friend", "mutable", "namespace", "new",
146 "noexcept", "not", "not_eq", "nullptr", "operator", "or", "or_eq",
147 "private", "protected", "public", "reinterpret_cast", "requires",
148 "static_assert", "static_cast", "template", "this", "throw",
149 "true", "try", "typeid", "typename", "using", "virtual", "xor",
150 "xor_eq", "override", "final",
151 ].iter().map(|s| s.to_string()).collect();
152
153 let operators = vec![
154 // Three-character.
155 "<<=", ">>=", "<=>",
156 // Two-character.
157 "->", "::", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
158 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
159 ".*", "->*",
160 ].iter().map(|s| s.to_string()).collect();
161
162 LangTokens {
163 keywords,
164 line_comment: "//".to_string(),
165 block_comment_open: "/*".to_string(),
166 block_comment_close: "*/".to_string(),
167 doc_comment_prefixes: vec!["///".to_string()],
168 operators,
169 string_delimiters: vec!['"'],
170 char_delimiter: Some('\''),
171 raw_string_prefix: "R\"".to_string(),
172 literal_prefixes: Vec::new(),
173 attribute_prefix: "[[".to_string(),
174 lifetime_prefix: None,
175 }
176}
177
178/// C# token definitions.
179pub fn csharp_tokens() -> LangTokens {
180 let keywords: BTreeSet<String> = [
181 "abstract", "as", "base", "bool", "break", "byte", "case", "catch",
182 "char", "checked", "class", "const", "continue", "decimal", "default",
183 "delegate", "do", "double", "else", "enum", "event", "explicit",
184 "extern", "false", "finally", "fixed", "float", "for", "foreach",
185 "goto", "if", "implicit", "in", "int", "interface", "internal",
186 "is", "lock", "long", "namespace", "new", "null", "object",
187 "operator", "out", "override", "params", "private", "protected",
188 "public", "readonly", "ref", "return", "sbyte", "sealed", "short",
189 "sizeof", "stackalloc", "static", "string", "struct", "switch",
190 "this", "throw", "true", "try", "typeof", "uint", "ulong",
191 "unchecked", "unsafe", "ushort", "using", "var", "virtual", "void",
192 "volatile", "while",
193 // Contextual keywords.
194 "async", "await", "get", "set", "value", "yield", "partial",
195 "where", "nameof", "when", "init", "record", "required",
196 ].iter().map(|s| s.to_string()).collect();
197
198 let operators = vec![
199 "<<=", ">>=",
200 "=>", "??", "?.", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
201 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
202 "??=", "..",
203 ].iter().map(|s| s.to_string()).collect();
204
205 LangTokens {
206 keywords,
207 line_comment: "//".to_string(),
208 block_comment_open: "/*".to_string(),
209 block_comment_close: "*/".to_string(),
210 doc_comment_prefixes: vec!["///".to_string()],
211 operators,
212 string_delimiters: vec!['"'],
213 char_delimiter: Some('\''),
214 raw_string_prefix: "@".to_string(),
215 literal_prefixes: Vec::new(),
216 attribute_prefix: "[".to_string(),
217 lifetime_prefix: None,
218 }
219}
220
221/// Go token definitions.
222pub fn go_tokens() -> LangTokens {
223 let keywords: BTreeSet<String> = [
224 "break", "case", "chan", "const", "continue", "default", "defer",
225 "else", "fallthrough", "for", "func", "go", "goto", "if",
226 "import", "interface", "map", "package", "range", "return",
227 "select", "struct", "switch", "type", "var",
228 ].iter().map(|s| s.to_string()).collect();
229
230 let operators = vec![
231 "<<=", ">>=",
232 ":=", "<-", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
233 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
234 "&^", "&^=", "...",
235 ].iter().map(|s| s.to_string()).collect();
236
237 LangTokens {
238 keywords,
239 line_comment: "//".to_string(),
240 block_comment_open: "/*".to_string(),
241 block_comment_close: "*/".to_string(),
242 doc_comment_prefixes: Vec::new(),
243 operators,
244 string_delimiters: vec!['"', '`'],
245 char_delimiter: Some('\''),
246 raw_string_prefix: String::new(),
247 literal_prefixes: Vec::new(),
248 attribute_prefix: String::new(),
249 lifetime_prefix: None,
250 }
251}
252
253/// JavaScript / TypeScript token definitions.
254pub fn js_tokens() -> LangTokens {
255 let keywords: BTreeSet<String> = [
256 "abstract", "arguments", "async", "await", "break", "case",
257 "catch", "class", "const", "continue", "debugger", "default",
258 "delete", "do", "else", "enum", "export", "extends", "false",
259 "finally", "for", "from", "function", "get", "if", "implements",
260 "import", "in", "instanceof", "interface", "let", "new", "null",
261 "of", "package", "private", "protected", "public", "return",
262 "set", "static", "super", "switch", "this", "throw", "true",
263 "try", "type", "typeof", "undefined", "var", "void", "while",
264 "with", "yield",
265 ].iter().map(|s| s.to_string()).collect();
266
267 let operators = vec![
268 ">>>", "<<=", ">>=",
269 "===", "!==", "**=", ">>>=",
270 "=>", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
271 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
272 "**", "??", "?.", "...",
273 ].iter().map(|s| s.to_string()).collect();
274
275 LangTokens {
276 keywords,
277 line_comment: "//".to_string(),
278 block_comment_open: "/*".to_string(),
279 block_comment_close: "*/".to_string(),
280 doc_comment_prefixes: vec!["/**".to_string()],
281 operators,
282 string_delimiters: vec!['"', '\'', '`'],
283 char_delimiter: None,
284 raw_string_prefix: String::new(),
285 literal_prefixes: Vec::new(),
286 attribute_prefix: "@".to_string(),
287 lifetime_prefix: None,
288 }
289}
290
291/// Java token definitions.
292pub fn java_tokens() -> LangTokens {
293 let keywords: BTreeSet<String> = [
294 "abstract", "assert", "boolean", "break", "byte", "case",
295 "catch", "char", "class", "const", "continue", "default", "do",
296 "double", "else", "enum", "extends", "false", "final", "finally",
297 "float", "for", "goto", "if", "implements", "import",
298 "instanceof", "int", "interface", "long", "native", "new",
299 "null", "package", "private", "protected", "public", "return",
300 "short", "static", "strictfp", "super", "switch", "synchronized",
301 "this", "throw", "throws", "transient", "true", "try", "void",
302 "volatile", "while",
303 ].iter().map(|s| s.to_string()).collect();
304
305 let operators = vec![
306 ">>>", "<<=", ">>=", ">>>=",
307 "->", "++", "--", "&&", "||", "==", "!=", "<=", ">=",
308 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
309 "::",
310 ].iter().map(|s| s.to_string()).collect();
311
312 LangTokens {
313 keywords,
314 line_comment: "//".to_string(),
315 block_comment_open: "/*".to_string(),
316 block_comment_close: "*/".to_string(),
317 doc_comment_prefixes: vec!["/**".to_string()],
318 operators,
319 string_delimiters: vec!['"'],
320 char_delimiter: Some('\''),
321 raw_string_prefix: String::new(),
322 literal_prefixes: Vec::new(),
323 attribute_prefix: "@".to_string(),
324 lifetime_prefix: None,
325 }
326}
327
328/// Python token definitions.
329pub fn python_tokens() -> LangTokens {
330 let keywords: BTreeSet<String> = [
331 "False", "None", "True", "and", "as", "assert", "async",
332 "await", "break", "class", "continue", "def", "del", "elif",
333 "else", "except", "finally", "for", "from", "global", "if",
334 "import", "in", "is", "lambda", "nonlocal", "not", "or",
335 "pass", "raise", "return", "try", "while", "with", "yield",
336 ].iter().map(|s| s.to_string()).collect();
337
338 let operators = vec![
339 "<<=", ">>=",
340 "**=", "//=",
341 "->", "==", "!=", "<=", ">=",
342 "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=", "<<", ">>",
343 "**", "//", ":=",
344 ].iter().map(|s| s.to_string()).collect();
345
346 LangTokens {
347 keywords,
348 line_comment: "#".to_string(),
349 block_comment_open: String::new(),
350 block_comment_close: String::new(),
351 doc_comment_prefixes: Vec::new(),
352 operators,
353 string_delimiters: vec!['"', '\''],
354 char_delimiter: None,
355 raw_string_prefix: "r".to_string(),
356 literal_prefixes: Vec::new(),
357 attribute_prefix: "@".to_string(),
358 lifetime_prefix: None,
359 }
360}
361
362/// Lex source text into a token stream.
363pub fn lex(src: &str, lang: &LangTokens) -> Outcome<Vec<Token>> {
364 let mut tokens = Vec::new();
365 let bytes = src.as_bytes();
366 let len = bytes.len();
367 let mut pos: usize = 0;
368
369 loop {
370 // Collect leading trivia.
371 let mut trivia: Vec<Trivia> = Vec::new();
372 loop {
373 if pos >= len {
374 break;
375 }
376 // Newline.
377 if bytes[pos] == b'\n' {
378 trivia.push(Trivia::Newline);
379 pos += 1;
380 continue;
381 }
382 if bytes[pos] == b'\r' && pos + 1 < len && bytes[pos + 1] == b'\n' {
383 trivia.push(Trivia::Newline);
384 pos += 2;
385 continue;
386 }
387 // Whitespace (not newline).
388 if bytes[pos] == b' ' || bytes[pos] == b'\t' {
389 let start = pos;
390 while pos < len && (bytes[pos] == b' ' || bytes[pos] == b'\t') {
391 pos += 1;
392 }
393 trivia.push(Trivia::Whitespace(src[start..pos].to_string()));
394 continue;
395 }
396 // Doc comments (must check before line comments).
397 // In Rust, `///` is a doc comment but `////` is a regular
398 // line comment. Reject prefixes that are followed by the
399 // same delimiter character (e.g. `/` after `///`).
400 let mut is_doc = false;
401 for prefix in &lang.doc_comment_prefixes {
402 if src[pos..].starts_with(prefix) {
403 let after = pos + prefix.len();
404 if !prefix.is_empty() && after < len {
405 let delim = prefix.as_bytes()[prefix.len() - 1];
406 if bytes[after] == delim {
407 continue;
408 }
409 }
410 let start = pos;
411 // Consume to end of line.
412 while pos < len && bytes[pos] != b'\n' {
413 pos += 1;
414 }
415 // Push as a doc-comment token, not trivia.
416 // We break out and handle it below.
417 // Actually, doc comments are tokens, not trivia.
418 // Rewind pos and break.
419 pos = start;
420 is_doc = true;
421 break;
422 }
423 }
424 if is_doc {
425 break;
426 }
427 // Line comment.
428 if !lang.line_comment.is_empty() && src[pos..].starts_with(&lang.line_comment) {
429 let start = pos;
430 while pos < len && bytes[pos] != b'\n' {
431 pos += 1;
432 }
433 trivia.push(Trivia::LineComment(src[start..pos].to_string()));
434 continue;
435 }
436 // Block comment.
437 if !lang.block_comment_open.is_empty() && src[pos..].starts_with(&lang.block_comment_open) {
438 let start = pos;
439 pos += lang.block_comment_open.len();
440 let mut depth = 1usize;
441 while pos < len && depth > 0 {
442 if src[pos..].starts_with(&lang.block_comment_close) {
443 depth -= 1;
444 pos += lang.block_comment_close.len();
445 } else if src[pos..].starts_with(&lang.block_comment_open) {
446 depth += 1;
447 pos += lang.block_comment_open.len();
448 } else {
449 pos = advance_char(src, pos);
450 }
451 }
452 trivia.push(Trivia::BlockComment(src[start..pos].to_string()));
453 continue;
454 }
455 break;
456 }
457
458 if pos >= len {
459 // EOF token.
460 tokens.push(Token {
461 kind: TokenKind::Eof,
462 text: String::new(),
463 leading_trivia: trivia,
464 span: Span { start: pos, end: pos },
465 });
466 break;
467 }
468
469 let tok_start = pos;
470
471 // Doc comment.
472 // A run of more slashes than the prefix is an ordinary comment,
473 // not documentation: Rust reads `////` as a plain comment and
474 // `///` as a doc comment, and treating the two alike moves a
475 // `////` line onto the item below it.
476 let doc_prefix = lang.doc_comment_prefixes.iter()
477 .find(|p| {
478 src[pos..].starts_with(p.as_str())
479 && !(p.ends_with('/')
480 && bytes.get(pos + p.len()) == Some(&b'/'))
481 })
482 .cloned();
483 if doc_prefix.is_some() {
484 let start = pos;
485 while pos < len && bytes[pos] != b'\n' {
486 pos += 1;
487 }
488 let text = src[start..pos].to_string();
489 tokens.push(Token {
490 kind: TokenKind::DocComment(text.clone()),
491 text,
492 leading_trivia: trivia,
493 span: Span { start: tok_start, end: pos },
494 });
495 continue;
496 }
497
498 // Attribute (e.g. #[...]).
499 if !lang.attribute_prefix.is_empty() && src[pos..].starts_with(&lang.attribute_prefix) {
500 let start = pos;
501 pos += lang.attribute_prefix.len();
502 let mut depth = 1usize;
503 while pos < len && depth > 0 {
504 match bytes[pos] {
505 b'[' => depth += 1,
506 b']' => depth -= 1,
507 _ => {}
508 }
509 if depth > 0 {
510 pos += 1;
511 } else {
512 pos += 1; // Consume the closing ].
513 }
514 }
515 tokens.push(Token {
516 kind: TokenKind::Attribute,
517 text: src[start..pos].to_string(),
518 leading_trivia: trivia,
519 span: Span { start: tok_start, end: pos },
520 });
521 continue;
522 }
523
524 // Prefixed literal (Rust `b"..."`, `br#"..."#`, `c"..."`,
525 // `b'x'`). Tried before the plain string and identifier paths,
526 // because the prefix would otherwise lex as its own identifier
527 // and the formatter would put a space between the two.
528 if let Some((end, kind)) = lex_prefixed_literal(src, bytes, pos, len, lang) {
529 let start = pos;
530 pos = end;
531 tokens.push(Token {
532 kind,
533 text: src[start..pos].to_string(),
534 leading_trivia: trivia,
535 span: Span { start: tok_start, end: pos },
536 });
537 continue;
538 }
539
540 // String literal.
541 if lang.string_delimiters.contains(&(bytes[pos] as char)) {
542 let delim = bytes[pos] as char;
543 let start = pos;
544 pos += 1;
545 while pos < len {
546 if bytes[pos] == b'\\' {
547 pos += 2; // Skip escaped character.
548 } else if bytes[pos] as char == delim {
549 pos += 1;
550 break;
551 } else {
552 pos += 1;
553 }
554 }
555 tokens.push(Token {
556 kind: TokenKind::StringLit,
557 text: src[start..pos].to_string(),
558 leading_trivia: trivia,
559 span: Span { start: tok_start, end: pos },
560 });
561 continue;
562 }
563
564 // Raw string (Rust r#"..."#).
565 if !lang.raw_string_prefix.is_empty()
566 && src[pos..].starts_with(&lang.raw_string_prefix)
567 && pos + lang.raw_string_prefix.len() < len
568 {
569 let after_r = pos + lang.raw_string_prefix.len();
570 let mut hashes = 0usize;
571 let mut p = after_r;
572 while p < len && bytes[p] == b'#' {
573 hashes += 1;
574 p += 1;
575 }
576 if p < len && bytes[p] == b'"' {
577 // Valid raw string opening.
578 let start = pos;
579 p += 1; // Skip opening ".
580 let close_pat: String =
581 std::iter::once('"').chain(std::iter::repeat('#').take(hashes)).collect();
582 while p < len {
583 if src[p..].starts_with(&close_pat) {
584 p += close_pat.len();
585 break;
586 }
587 p = advance_char(src, p);
588 }
589 pos = p;
590 tokens.push(Token {
591 kind: TokenKind::StringLit,
592 text: src[start..pos].to_string(),
593 leading_trivia: trivia,
594 span: Span { start: tok_start, end: pos },
595 });
596 continue;
597 }
598 }
599
600 // Character literal.
601 if let Some(cd) = lang.char_delimiter {
602 if bytes[pos] as char == cd {
603 // Distinguish char literal from lifetime:
604 // char literal: 'a', '\n', etc.
605 // lifetime: 'a followed by an ident-continue character.
606 let start = pos;
607 pos += 1;
608 if pos < len && bytes[pos] == b'\\' {
609 // Escaped char literal.
610 pos += 1;
611 while pos < len && bytes[pos] as char != cd {
612 pos += 1;
613 }
614 if pos < len { pos += 1; }
615 tokens.push(Token {
616 kind: TokenKind::CharLit,
617 text: src[start..pos].to_string(),
618 leading_trivia: trivia,
619 span: Span { start: tok_start, end: pos },
620 });
621 continue;
622 } else if pos < len && pos + 1 < len && bytes[pos + 1] as char == cd {
623 // Single-char literal like 'a'.
624 pos += 2;
625 tokens.push(Token {
626 kind: TokenKind::CharLit,
627 text: src[start..pos].to_string(),
628 leading_trivia: trivia,
629 span: Span { start: tok_start, end: pos },
630 });
631 continue;
632 } else if pos < len && is_ident_start(bytes[pos]) {
633 // Lifetime.
634 let _lstart = pos;
635 while pos < len && is_ident_continue(bytes[pos]) {
636 pos += 1;
637 }
638 tokens.push(Token {
639 kind: TokenKind::Lifetime,
640 text: src[start..pos].to_string(),
641 leading_trivia: trivia,
642 span: Span { start: tok_start, end: pos },
643 });
644 continue;
645 } else {
646 // Bare apostrophe — treat as punctuation.
647 pos = start + 1;
648 tokens.push(Token {
649 kind: TokenKind::Punct('\''),
650 text: "'".to_string(),
651 leading_trivia: trivia,
652 span: Span { start: tok_start, end: pos },
653 });
654 continue;
655 }
656 }
657 }
658
659 // Number.
660 if bytes[pos].is_ascii_digit() {
661 let start = pos;
662 // Consume digits, hex prefix, underscores, dots, exponent.
663 while pos < len && (bytes[pos].is_ascii_alphanumeric()
664 || bytes[pos] == b'_'
665 || bytes[pos] == b'.'
666 || bytes[pos] == b'+'
667 || bytes[pos] == b'-')
668 {
669 // Avoid consuming a `..` range operator.
670 if bytes[pos] == b'.' && pos + 1 < len && bytes[pos + 1] == b'.' {
671 break;
672 }
673 // A `.` followed by an identifier start is a field or
674 // tuple-index access dot (e.g. `0.connect_options`,
675 // `1.5.method`), not a decimal point — stop before it.
676 if bytes[pos] == b'.'
677 && pos + 1 < len
678 && is_ident_start(bytes[pos + 1])
679 {
680 break;
681 }
682 // Avoid consuming `+`/`-` unless it's part of an exponent.
683 if (bytes[pos] == b'+' || bytes[pos] == b'-')
684 && pos > start
685 && bytes[pos - 1] != b'e'
686 && bytes[pos - 1] != b'E'
687 {
688 break;
689 }
690 pos += 1;
691 }
692 tokens.push(Token {
693 kind: TokenKind::Number,
694 text: src[start..pos].to_string(),
695 leading_trivia: trivia,
696 span: Span { start: tok_start, end: pos },
697 });
698 continue;
699 }
700
701 // Identifier / keyword.
702 if is_ident_start(bytes[pos]) {
703 let start = pos;
704 while pos < len && is_ident_continue(bytes[pos]) {
705 pos += 1;
706 }
707 let word = &src[start..pos];
708 // Check for macro invocation (ident followed by `!`).
709 let kind = if pos < len && bytes[pos] == b'!' && !lang.keywords.contains(word) {
710 // Don't consume the `!` here, let it be a separate punct token
711 // unless it's a macro name.
712 // Actually, consume it as part of the macro name.
713 // pos += 1; // No — the `!` is part of the macro syntax, not the name.
714 if lang.keywords.contains(word) {
715 TokenKind::Keyword(word.to_string())
716 } else {
717 TokenKind::Ident
718 }
719 } else if lang.keywords.contains(word) {
720 TokenKind::Keyword(word.to_string())
721 } else {
722 TokenKind::Ident
723 };
724 tokens.push(Token {
725 kind,
726 text: word.to_string(),
727 leading_trivia: trivia,
728 span: Span { start: tok_start, end: pos },
729 });
730 continue;
731 }
732
733 // Multi-character operator.
734 let matched_op = lang.operators.iter()
735 .find(|op| src[pos..].starts_with(op.as_str()))
736 .cloned();
737 if let Some(op) = matched_op {
738 pos += op.len();
739 tokens.push(Token {
740 kind: TokenKind::Operator(op.clone()),
741 text: op,
742 leading_trivia: trivia,
743 span: Span { start: tok_start, end: pos },
744 });
745 continue;
746 }
747
748 // Single-character punctuation (or full multi-byte char).
749 let ch = src[pos..].chars().next().unwrap_or('\0');
750 let next = advance_char(src, pos);
751 tokens.push(Token {
752 kind: TokenKind::Punct(ch),
753 text: src[pos..next].to_string(),
754 leading_trivia: trivia,
755 span: Span { start: tok_start, end: next },
756 });
757 pos = next;
758 }
759
760 Ok(tokens)
761}
762
763/// Advance past one complete UTF-8 character.
764fn advance_char(src: &str, pos: usize) -> usize {
765 let mut p = pos + 1;
766 while p < src.len() && !src.is_char_boundary(p) {
767 p += 1;
768 }
769 p
770}
771
772/// Where a prefixed literal beginning at `pos` ends, and what it is.
773///
774/// A prefix only counts when a literal actually follows it, so `brand`
775/// stays an identifier and Rust's raw identifier `r#type` is left to
776/// the raw-string path, which already rejects it.
777fn lex_prefixed_literal(
778 src: &str,
779 bytes: &[u8],
780 pos: usize,
781 len: usize,
782 lang: &LangTokens,
783) -> Option<(usize, TokenKind)> {
784 for prefix in &lang.literal_prefixes {
785 if !src[pos..].starts_with(prefix.as_str()) {
786 continue;
787 }
788 let after = pos + prefix.len();
789 if after >= len {
790 continue;
791 }
792 // A prefix is only a prefix when an ident character does not
793 // follow the literal opener, e.g. `b"x"` but not `brand`.
794 match bytes[after] {
795 b'"' => {
796 let mut p = after + 1;
797 while p < len {
798 if bytes[p] == b'\\' {
799 p += 2;
800 } else if bytes[p] == b'"' {
801 p += 1;
802 break;
803 } else {
804 p += 1;
805 }
806 }
807 return Some((p.min(len), TokenKind::StringLit));
808 }
809 b'#' => {
810 let mut hashes = 0usize;
811 let mut p = after;
812 while p < len && bytes[p] == b'#' {
813 hashes += 1;
814 p += 1;
815 }
816 if p >= len || bytes[p] != b'"' {
817 continue; // Not a raw literal after all.
818 }
819 p += 1;
820 let close: String = std::iter::once('"')
821 .chain(std::iter::repeat('#').take(hashes))
822 .collect();
823 while p < len {
824 if src[p..].starts_with(close.as_str()) {
825 p += close.len();
826 break;
827 }
828 p = advance_char(src, p);
829 }
830 return Some((p.min(len), TokenKind::StringLit));
831 }
832 _ => {
833 // Byte character literal, e.g. `b'a'` or `b'\n'`.
834 if lang.char_delimiter == Some(bytes[after] as char) {
835 let mut p = after + 1;
836 while p < len {
837 if bytes[p] == b'\\' {
838 p += 2;
839 } else if lang.char_delimiter == Some(bytes[p] as char) {
840 p += 1;
841 break;
842 } else {
843 p += 1;
844 }
845 }
846 return Some((p.min(len), TokenKind::CharLit));
847 }
848 }
849 }
850 }
851 None
852}
853
854fn is_ident_start(b: u8) -> bool {
855 b.is_ascii_alphabetic() || b == b'_'
856}
857
858fn is_ident_continue(b: u8) -> bool {
859 b.is_ascii_alphanumeric() || b == b'_'
860}
861
862
863#[cfg(test)]
864mod tests {
865 use super::*;
866
867 fn lex_rust(src: &str) -> Vec<Token> {
868 let lang = rust_tokens();
869 lex(src, &lang).expect("lex failed")
870 }
871
872 #[test]
873 fn test_simple_fn() {
874 let tokens = lex_rust("fn main() {}");
875 // fn, main, (, ), {, }, EOF
876 assert_eq!(tokens.len(), 7);
877 assert_eq!(tokens[0].kind, TokenKind::Keyword("fn".into()));
878 assert_eq!(tokens[1].kind, TokenKind::Ident);
879 assert_eq!(tokens[1].text, "main");
880 assert_eq!(tokens[2].kind, TokenKind::Punct('('));
881 assert_eq!(tokens[3].kind, TokenKind::Punct(')'));
882 assert_eq!(tokens[4].kind, TokenKind::Punct('{'));
883 assert_eq!(tokens[5].kind, TokenKind::Punct('}'));
884 assert_eq!(tokens[6].kind, TokenKind::Eof);
885 }
886
887 #[test]
888 fn test_trivia_preserved() {
889 let tokens = lex_rust("fn main() {\n // hi\n}");
890 // fn has no leading trivia.
891 assert!(tokens[0].leading_trivia.is_empty());
892 // main has whitespace trivia.
893 assert_eq!(tokens[1].leading_trivia.len(), 1);
894 match &tokens[1].leading_trivia[0] {
895 Trivia::Whitespace(s) => assert_eq!(s, " "),
896 other => panic!("expected whitespace, got {:?}", other),
897 }
898 }
899
900 #[test]
901 fn test_string_literal() {
902 let tokens = lex_rust(r#"let s = "hello \"world\"";"#);
903 // let, s, =, "hello \"world\"", ;, EOF
904 let string_tok = tokens.iter().find(|t| matches!(t.kind, TokenKind::StringLit)).unwrap();
905 assert_eq!(string_tok.text, r#""hello \"world\"""#);
906 }
907
908 #[test]
909 fn test_operators() {
910 let tokens = lex_rust("a -> b => c :: d");
911 let ops: Vec<&str> = tokens.iter().filter_map(|t| {
912 if let TokenKind::Operator(ref s) = t.kind { Some(s.as_str()) } else { None }
913 }).collect();
914 assert_eq!(ops, vec!["->", "=>", "::"]);
915 }
916
917 #[test]
918 fn test_doc_comment() {
919 let tokens = lex_rust("/// A doc comment.\nfn foo() {}");
920 assert!(matches!(tokens[0].kind, TokenKind::DocComment(_)));
921 assert_eq!(tokens[0].text, "/// A doc comment.");
922 }
923
924 #[test]
925 fn test_attribute() {
926 let tokens = lex_rust("#[derive(Clone, Debug)]\nstruct Foo;");
927 assert_eq!(tokens[0].kind, TokenKind::Attribute);
928 assert_eq!(tokens[0].text, "#[derive(Clone, Debug)]");
929 }
930
931 #[test]
932 fn test_lifetime() {
933 let tokens = lex_rust("fn foo<'a>(x: &'a str) {}");
934 let lifetimes: Vec<&str> = tokens.iter().filter_map(|t| {
935 if matches!(t.kind, TokenKind::Lifetime) { Some(t.text.as_str()) } else { None }
936 }).collect();
937 assert_eq!(lifetimes, vec!["'a", "'a"]);
938 }
939
940 #[test]
941 fn test_char_literal() {
942 let tokens = lex_rust("let c = 'x';");
943 let char_tok = tokens.iter().find(|t| matches!(t.kind, TokenKind::CharLit)).unwrap();
944 assert_eq!(char_tok.text, "'x'");
945 }
946
947 #[test]
948 fn test_number() {
949 let tokens = lex_rust("let n = 42_000;");
950 let num = tokens.iter().find(|t| matches!(t.kind, TokenKind::Number)).unwrap();
951 assert_eq!(num.text, "42_000");
952 }
953
954 #[test]
955 fn test_block_comment() {
956 let tokens = lex_rust("/* block */ fn foo() {}");
957 // The block comment is leading trivia on `fn`.
958 assert_eq!(tokens[0].kind, TokenKind::Keyword("fn".into()));
959 assert!(tokens[0].leading_trivia.iter().any(|t| matches!(t, Trivia::BlockComment(_))));
960 }
961
962 #[test]
963 fn test_line_comment_trivia() {
964 let tokens = lex_rust("// line comment\nfn foo() {}");
965 // Line comment is leading trivia on `fn`.
966 assert_eq!(tokens[0].kind, TokenKind::Keyword("fn".into()));
967 assert!(tokens[0].leading_trivia.iter().any(|t| matches!(t, Trivia::LineComment(_))));
968 }
969
970 #[test]
971 fn test_lex_c() {
972 let lang = c_tokens();
973 let tokens = lex("int main() { return 0; }", &lang).expect("lex failed");
974 assert_eq!(tokens[0].kind, TokenKind::Keyword("int".into()));
975 assert_eq!(tokens[1].kind, TokenKind::Ident);
976 assert_eq!(tokens[1].text, "main");
977 }
978
979 #[test]
980 fn test_lex_go() {
981 let lang = go_tokens();
982 let tokens = lex("func main() { fmt.Println(\"hello\") }", &lang).expect("lex failed");
983 assert_eq!(tokens[0].kind, TokenKind::Keyword("func".into()));
984 assert_eq!(tokens[1].kind, TokenKind::Ident);
985 assert_eq!(tokens[1].text, "main");
986 }
987
988 #[test]
989 fn test_lex_js() {
990 let lang = js_tokens();
991 let tokens = lex("const x = async () => { await fetch(); };", &lang).expect("lex failed");
992 assert_eq!(tokens[0].kind, TokenKind::Keyword("const".into()));
993 let arrow = tokens.iter().find(|t| matches!(&t.kind, TokenKind::Operator(op) if op == "=>"));
994 assert!(arrow.is_some(), "expected => operator");
995 }
996
997 #[test]
998 fn test_lex_java() {
999 let lang = java_tokens();
1000 let tokens = lex("public class Foo { void bar() {} }", &lang).expect("lex failed");
1001 assert_eq!(tokens[0].kind, TokenKind::Keyword("public".into()));
1002 assert_eq!(tokens[1].kind, TokenKind::Keyword("class".into()));
1003 assert_eq!(tokens[2].kind, TokenKind::Ident);
1004 assert_eq!(tokens[2].text, "Foo");
1005 }
1006
1007 #[test]
1008 fn test_lex_python() {
1009 let lang = python_tokens();
1010 let tokens = lex("def foo(x, y):\n return x + y", &lang).expect("lex failed");
1011 assert_eq!(tokens[0].kind, TokenKind::Keyword("def".into()));
1012 assert_eq!(tokens[1].kind, TokenKind::Ident);
1013 assert_eq!(tokens[1].text, "foo");
1014 }
1015
1016 #[test]
1017 fn test_lex_python_comment() {
1018 let lang = python_tokens();
1019 let tokens = lex("# a comment\nx = 1", &lang).expect("lex failed");
1020 // Comment should be trivia on `x`.
1021 assert_eq!(tokens[0].kind, TokenKind::Ident);
1022 assert!(tokens[0].leading_trivia.iter().any(|t| matches!(t, Trivia::LineComment(_))));
1023 }
1024
1025 #[test]
1026 fn test_nested_generics() {
1027 let tokens = lex_rust("Outcome<Vec<Token>>");
1028 let texts: Vec<&str> = tokens.iter()
1029 .filter(|t| !matches!(t.kind, TokenKind::Eof))
1030 .map(|t| t.text.as_str()).collect();
1031 // Should be: Outcome, <, Vec, <, Token, >>, not >>
1032 println!("tokens: {:?}", texts);
1033 // The >> should be lexed as the >> operator, but in a type
1034 // context it's two closing >. This is a known ambiguity.
1035 // For formatting, we just need to know it's there.
1036 }
1037
1038 #[test]
1039 fn test_range_vs_dot() {
1040 let tokens = lex_rust("0..10");
1041 // 0, .., 10, EOF
1042 assert_eq!(tokens[0].kind, TokenKind::Number);
1043 assert_eq!(tokens[0].text, "0");
1044 assert_eq!(tokens[1].kind, TokenKind::Operator("..".into()));
1045 assert_eq!(tokens[2].kind, TokenKind::Number);
1046 assert_eq!(tokens[2].text, "10");
1047 }
1048}