Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_ore/src/fastexport.rs

61.1 KiB, 196 runs

created by r1870400018:17869, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! A parser for git's fast-import stream, the format `git fast-export` emits.
2//!
3//! Git keeps one documented interface for foreign consumers of a repository:
4//! the byte stream described by `git-fast-import(1)`. Everything else about
5//! how git stores history -- packfiles, the reference backend, the object hash
6//! -- is internal and has changed more than once. Reading the stream instead of
7//! the repository means none of that reaches this code. The Fossil version
8//! control system imports git the same way, and for the same reason.
9//!
10//! # A pure parser
11//!
12//! Bytes in, events out. Nothing here spawns a process, opens a file or reads
13//! a clock: obtaining the stream is the caller's job, whether that is a pipe
14//! from `git fast-export`, a file, or a socket. That keeps the module portable
15//! to `wasm32-unknown-unknown` along with the rest of the crate, and it keeps
16//! it testable against hand-written streams.
17//!
18//! # Incremental
19//!
20//! [`Parser::feed`] takes whatever bytes have arrived, and
21//! [`Parser::next_event`] yields events as they complete, so a whole repository
22//! never has to be held in memory. Memory is bounded by the largest single
23//! object in the stream rather than by the stream: one blob payload, or one
24//! commit with its file changes, is assembled before it is handed over. Feeding
25//! a stream one byte at a time and feeding it all at once produce exactly the
26//! same events.
27//!
28//! # Example
29//!
30//! ```
31//! use oxedyne_fe2o3_core::prelude::*;
32//! use oxedyne_fe2o3_ore::fastexport::{Event, Parser};
33//!
34//! # fn main() -> Outcome<()> {
35//! let mut parser = Parser::new();
36//! parser.feed(b"blob\nmark :1\ndata 5\nhello\n");
37//! parser.end();
38//! while let Some(event) = res!(parser.next_event()) {
39//! match event {
40//! Event::Blob(blob) => assert_eq!(blob.data, b"hello"),
41//! other => return Err(err!("Unexpected event {:?}.", other; Test)),
42//! }
43//! }
44//! # Ok(())
45//! # }
46//! ```
47//!
48//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
49//! Anthropic Claude
50
51use crate::op::Mode;
52
53use oxedyne_fe2o3_core::prelude::*;
54
55use std::fmt;
56
57
58const COMPACT_THRESHOLD: usize = 64 * 1024; // consumed prefix tolerated
59const SHOW_LIMIT: usize = 120; // bytes of a line quoted in an error
60
61
62// ---------------------------------------------------------------------------
63// Stream vocabulary.
64// ---------------------------------------------------------------------------
65
66/// A reference to an object, as the stream is able to spell one.
67///
68/// The stream does not distinguish an object identifier from a branch name or
69/// a revision expression, and a parser that has not read the repository cannot
70/// tell them apart either, so both arrive as [`ObjRef::Name`]. Only a mark,
71/// which the stream mints itself, is unambiguous.
72#[derive(Clone, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
73pub enum ObjRef {
74 Mark(u64), // `:N`
75 Name(String), // hex oid, branch name, or revision expression
76}
77
78impl fmt::Display for ObjRef {
79 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
80 match self {
81 Self::Mark(n) => write!(f, ":{}", n),
82 Self::Name(name) => write!(f, "{}", name),
83 }
84 }
85}
86
87
88/// Where the content of a file entry comes from.
89///
90/// The stream either points at an object it has already described, or carries
91/// the bytes in line with the file command.
92#[derive(Clone, Debug, Eq, PartialEq)]
93pub enum BlobRef {
94 Mark(u64), // `:N`
95 Name(String), // hex oid or revision expression
96 Inline(Vec<u8>), // bytes that followed the file command as a `data` payload
97}
98
99
100/// The file modes git records in a tree.
101///
102/// Git stores a small fixed set, so this is an enum and not a number: a mode
103/// outside the set is a stream this parser will not guess at.
104#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
105pub enum FileMode {
106 Normal, // 100644
107 Executable, // 100755
108 Symlink, // 120000
109 Gitlink, // 160000, a commit of another repository
110 Subdirectory, // 040000, a directory entry
111}
112
113impl FileMode {
114 /// Accepts both the six digit form and the shortened form git also permits.
115 pub fn from_bytes(bytes: &[u8])
116 -> Outcome<Self>
117 {
118 match bytes {
119 b"100644" | b"644" => Ok(Self::Normal),
120 b"100755" | b"755" => Ok(Self::Executable),
121 b"120000" => Ok(Self::Symlink),
122 b"160000" => Ok(Self::Gitlink),
123 b"040000" | b"40000" => Ok(Self::Subdirectory),
124 other => Err(err!(
125 "File mode {:?} is not one of the modes git records.", show(other);
126 Decode, Input, Invalid)),
127 }
128 }
129
130 /// `None` is a gitlink or a directory entry: both are things a tree can point
131 /// at and neither is a file with bytes, so there is nothing for
132 /// [`crate::op::Op::FileMode`] to say about them. A consumer that meets one
133 /// has a decision to make -- refuse, or leave the entry out -- and it is the
134 /// consumer that knows which path and which commit to name, so it makes that
135 /// decision rather than being handed a guess.
136 pub const fn as_op_mode(&self) -> Option<Mode> {
137 match self {
138 Self::Normal => Some(Mode::Normal),
139 Self::Executable => Some(Mode::Executable),
140 Self::Symlink => Some(Mode::Symlink),
141 Self::Gitlink | Self::Subdirectory => None,
142 }
143 }
144
145 pub const fn as_str(&self) -> &'static str {
146 match self {
147 Self::Normal => "100644",
148 Self::Executable => "100755",
149 Self::Symlink => "120000",
150 Self::Gitlink => "160000",
151 Self::Subdirectory => "040000",
152 }
153 }
154}
155
156impl fmt::Display for FileMode {
157 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
158 write!(f, "{}", self.as_str())
159 }
160}
161
162
163/// An offset from UTC, as an identity line carries it.
164#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
165pub struct TzOffset {
166 pub mins: i32, // minutes, positive east of UTC
167 pub neg: bool, // minus as written; only `-0000`, the unknown zone, needs it
168}
169
170impl TzOffset {
171 pub const fn new(mins: i32) -> Self {
172 Self { mins, neg: mins < 0 }
173 }
174
175 pub const fn secs(&self) -> i32 {
176 self.mins * 60
177 }
178}
179
180impl fmt::Display for TzOffset {
181 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
182 let mag = self.mins.abs();
183 write!(f, "{}{:02}{:02}", if self.neg { '-' } else { '+' }, mag / 60, mag % 60)
184 }
185}
186
187
188/// The moment an identity line records.
189#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
190pub struct When {
191 pub secs: i64, // since the Unix epoch, which git allows to be negative
192 pub tz: TzOffset,
193}
194
195impl fmt::Display for When {
196 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
197 write!(f, "{} {}", self.secs, self.tz)
198 }
199}
200
201
202/// One `author`, `committer` or `tagger` line.
203///
204/// The name and the email arrive as bytes because git does not require either
205/// to be UTF-8, and a repository old enough to be worth importing is exactly
206/// the kind that has a Latin-1 name in it.
207#[derive(Clone, Debug, Eq, Hash, Ord, PartialEq, PartialOrd)]
208pub struct Person {
209 pub name: Vec<u8>, // may be empty
210 pub email: Vec<u8>, // without its angle brackets
211 pub when: When,
212}
213
214impl Person {
215 pub fn name_lossy(&self) -> String {
216 String::from_utf8_lossy(&self.name).into_owned()
217 }
218
219 pub fn email_lossy(&self) -> String {
220 String::from_utf8_lossy(&self.email).into_owned()
221 }
222}
223
224
225/// Splits git's `<seconds> <offset>` off the end of an identity line.
226///
227/// The line is git's own -- `Jason Hoogland <hoogland@gmail.com> 1735089438 +0800`
228/// -- and this is the one place that decides where the name ends and the moment
229/// begins. It is here rather than in each reader because the readers cannot be
230/// allowed to disagree: [`crate::op::AUTHOR_TRAILER`] carries such a line inside a
231/// mark, a mirror reads it to author a commit under the name it arrived with, and
232/// a forge reads it to show a person that name. Two implementations agreeing
233/// because they were written to agree is the arrangement this project has paid for
234/// before.
235///
236/// # All of it or none of it
237///
238/// A tail that is not a moment leaves the whole line as the identity, and this is
239/// the case worth stating: `J H <j@h.test>` splits on spaces perfectly well and
240/// would come back as `J` under a rule that only counted fields. So the tail is
241/// handed to [`parse_when`], the one reader of git's raw date format here, and its
242/// refusal is the answer -- an error is what "there is no moment on the end of
243/// this" looks like, rather than something being wrong.
244///
245/// A negative second is a commit dated before 1970, which git permits and which an
246/// import writes back as it read it.
247///
248/// ```
249/// use oxedyne_fe2o3_ore::fastexport::split_identity_line;
250///
251/// let (who, when) = match split_identity_line(
252/// "Jason Hoogland <hoogland@gmail.com> 1735089438 +0800")
253/// {
254/// Some(split) => split,
255/// None => panic!("a whole identity line splits"),
256/// };
257/// assert_eq!(who, "Jason Hoogland <hoogland@gmail.com>");
258/// assert_eq!(when.secs, 1735089438);
259/// assert_eq!(format!("{}", when.tz), "+0800");
260///
261/// // A tail that is not a moment, however much it looks like one. Losing
262/// // somebody over a malformed offset would defeat the point of carrying the
263/// // name at all, so all of the line is the person.
264/// for line in [
265/// "J H <j@h.test>",
266/// "J H <j@h.test> 1735089438 tomorrow",
267/// "J H <j@h.test> whenever +0800",
268/// "J H <j@h.test> +0800",
269/// "J H <j@h.test> 1735089438 +08:00",
270/// "replica 4256968235 <4256968235@replica.invalid>",
271/// ] {
272/// assert!(split_identity_line(line).is_none(), "{:?} ends in no moment", line);
273/// }
274///
275/// // A commit dated before 1970, which git writes with a negative second, in a
276/// // zone that is not a whole hour.
277/// let (who, when) = match split_identity_line("J H <j@h.test> -14182940 -0330") {
278/// Some(split) => split,
279/// None => panic!("that is a moment"),
280/// };
281/// assert_eq!(who, "J H <j@h.test>");
282/// assert_eq!(when.secs, -14182940);
283/// assert_eq!(format!("{}", when.tz), "-0330");
284/// ```
285pub fn split_identity_line(line: &str) -> Option<(&str, When)> {
286 let offset = match line.rfind(' ') {
287 Some(at) => at,
288 None => return None,
289 };
290 let secs = match line[..offset].rfind(' ') {
291 Some(at) => at,
292 None => return None,
293 };
294 match parse_when(line[secs + 1..].as_bytes(), "identity") {
295 Ok(when) => Some((&line[..secs], when)),
296 Err(_) => None,
297 }
298}
299
300/// Returns the identity an identity line names, which is all of it where the tail
301/// is not a moment.
302///
303/// What a reader showing a person a name wants: the moment is bookkeeping, and a
304/// page printing the value whole prints a timestamp in the middle of somebody's
305/// name.
306pub fn identity_in(line: &str) -> &str {
307 match split_identity_line(line) {
308 Some((identity, _)) => identity,
309 None => line,
310 }
311}
312
313
314/// A commit signature carried verbatim through the stream.
315#[derive(Clone, Debug, Eq, PartialEq)]
316pub struct GpgSig {
317 pub algo: String, // `sha1` or `sha256`
318 pub format: String, // `openpgp`, `x509`, `ssh`
319 pub sig: Vec<u8>, // unexamined
320}
321
322
323/// One change a commit makes to the tree.
324///
325/// Paths are bytes: git permits any byte but NUL and the path separator in a
326/// path component, and the stream's C-style quoting exists precisely so that
327/// such paths survive. They are unquoted here, so what a consumer receives is
328/// the path itself.
329#[derive(Clone, Debug, Eq, PartialEq)]
330pub enum FileChange {
331 Modify { // `M`, sets the content and mode of a path
332 mode: FileMode,
333 data: BlobRef,
334 path: Vec<u8>,
335 },
336 Delete { // `D`
337 path: Vec<u8>,
338 },
339 Copy { // `C`, leaving the source in place
340 src: Vec<u8>,
341 dst: Vec<u8>,
342 },
343 Rename { // `R`
344 src: Vec<u8>,
345 dst: Vec<u8>,
346 },
347 DeleteAll, // `deleteall`, emptying the tree before the rest apply
348 Note { // `N`, annotating a commit
349 data: BlobRef,
350 commit: ObjRef,
351 },
352}
353
354
355/// A `blob` command and its payload.
356#[derive(Clone, Debug, Eq, PartialEq)]
357pub struct Blob {
358 pub mark: Option<u64>,
359 pub original_oid: Option<String>, // its name in the exporting repository
360 // The payload is never line-parsed, so a blob holding NUL bytes or bare
361 // line feeds survives untouched.
362 pub data: Vec<u8>,
363}
364
365
366/// A `commit` command, complete with its file changes.
367#[derive(Clone, Debug, Eq, PartialEq)]
368pub struct Commit {
369 pub refname: String,
370 pub mark: Option<u64>,
371 pub original_oid: Option<String>,
372 pub author: Option<Person>, // absent where it is the committer
373 pub committer: Person,
374 pub encoding: Option<Vec<u8>>, // of the message
375 pub gpgsig: Option<GpgSig>,
376 pub message: Vec<u8>,
377 // Parents. `from` is absent for a root commit, and where the branch
378 // already stands where the commit builds on.
379 pub from: Option<ObjRef>,
380 pub merges: Vec<ObjRef>, // the rest, in order
381 pub changes: Vec<FileChange>, // stream order is apply order
382}
383
384
385/// A `tag` command.
386#[derive(Clone, Debug, Eq, PartialEq)]
387pub struct Tag {
388 pub name: String, // short, without the `refs/tags/` prefix
389 pub mark: Option<u64>,
390 pub original_oid: Option<String>,
391 pub from: ObjRef, // the object tagged
392 pub tagger: Option<Person>, // an older repository's stream may omit it
393 pub message: Vec<u8>,
394}
395
396
397/// One complete command from the stream.
398///
399/// A commit arrives whole, with its file changes attached, because a consumer
400/// has to see the whole commit before it can apply any of it. A blob arrives
401/// whole for the same reason: its payload is what the consumer wanted.
402#[derive(Clone, Debug, Eq, PartialEq)]
403pub enum Event {
404 Blob(Blob),
405 Commit(Commit),
406 Tag(Tag),
407 // A `reset` points a reference somewhere. Without `from` the reference
408 // is only declared, and the commit that follows defines it.
409 Reset {
410 refname: String,
411 from: Option<ObjRef>,
412 },
413 Progress(Vec<u8>), // text for the operator; means nothing to the import
414 Checkpoint, // asks the consumer to make its work durable
415 Alias { // a second mark for an object already named
416 mark: u64,
417 to: ObjRef,
418 },
419 Feature { // what the stream requires of its consumer
420 name: String,
421 arg: Option<String>,
422 },
423 Opt(Vec<u8>), // an `option`, which a consumer may ignore
424 Done, // the stream is complete, not merely cut off
425}
426
427
428// ---------------------------------------------------------------------------
429// The parser.
430// ---------------------------------------------------------------------------
431
432/// An incremental parser for the fast-import stream.
433///
434/// Bytes go in through [`Parser::feed`], events come out through
435/// [`Parser::next_event`], and [`Parser::end`] declares that no more bytes are
436/// coming. Until `end` has been called, `next_event` returning `None` means
437/// only that the next command is not yet complete; afterwards it means the
438/// stream is finished.
439#[derive(Debug, Default)]
440pub struct Parser {
441 buf: Vec<u8>, // fed but not yet turned into events, prefix included
442 pos: usize, // how much of `buf` has been consumed
443 eof: bool,
444}
445
446impl Parser {
447 pub fn new() -> Self {
448 Self::default()
449 }
450
451 /// Chunk boundaries carry no meaning: a command may be split anywhere, and
452 /// the same stream delivered in different chunkings yields the same
453 /// events.
454 ///
455 /// A command that arrives split is parsed again from its first byte once
456 /// more input lands, so the work done per chunk is bounded by the size of
457 /// one command and not by the stream. Feeding whole reads rather than
458 /// single bytes keeps that cost where it belongs.
459 pub fn feed(&mut self, chunk: &[u8]) {
460 self.buf.extend_from_slice(chunk);
461 }
462
463 /// After this, a command that is still incomplete is an error rather than
464 /// a request for more input, which is how a truncated stream is caught.
465 pub fn end(&mut self) {
466 self.eof = true;
467 }
468
469 /// Has the stream ended and every byte of it become an event?
470 pub fn is_exhausted(&self) -> bool {
471 self.eof && self.remaining().iter().all(|b| *b == b'\n')
472 }
473
474 pub fn remaining(&self) -> &[u8] {
475 &self.buf[self.pos..]
476 }
477
478 /// `None` means that the next command is not yet complete, or, once
479 /// [`Parser::end`] has been called, that the stream is finished. An error
480 /// names the offending line.
481 pub fn next_event(&mut self)
482 -> Outcome<Option<Event>>
483 {
484 // Blank lines separate commands and carry nothing.
485 while self.pos < self.buf.len() && self.buf[self.pos] == b'\n' {
486 self.pos += 1;
487 }
488 self.compact();
489 if self.pos >= self.buf.len() {
490 return Ok(None);
491 }
492 match res!(parse_command(&self.buf[self.pos..], self.eof)) {
493 Some((event, used)) => {
494 self.pos += used;
495 self.compact();
496 Ok(Some(event))
497 },
498 None => {
499 if self.eof {
500 Err(err!(
501 "The stream ends part way through a command beginning \
502 \"{}\".", show(first_line(&self.buf[self.pos..]));
503 Decode, Input, Missing))
504 } else {
505 Ok(None)
506 }
507 },
508 }
509 }
510
511 /// Drops the consumed prefix of the buffer once it is worth the move.
512 fn compact(&mut self) {
513 if self.pos == self.buf.len() {
514 self.buf.clear();
515 self.pos = 0;
516 } else if self.pos >= COMPACT_THRESHOLD {
517 self.buf.drain(..self.pos);
518 self.pos = 0;
519 }
520 }
521}
522
523/// The convenience form of [`Parser`] for a stream small enough to have
524/// already been read, used mostly in tests.
525pub fn parse_all(bytes: &[u8])
526 -> Outcome<Vec<Event>>
527{
528 let mut parser = Parser::new();
529 parser.feed(bytes);
530 parser.end();
531 let mut events = Vec::new();
532 while let Some(event) = res!(parser.next_event()) {
533 events.push(event);
534 }
535 Ok(events)
536}
537
538
539// ---------------------------------------------------------------------------
540// Line reading.
541// ---------------------------------------------------------------------------
542
543/// The result of asking for the next line.
544enum Line<'a> {
545 Got(&'a [u8], usize), // the line, and the offset just past its line feed
546 End, // no bytes remain and none are coming
547 Need, // more input is needed before the line can be read
548}
549
550/// Reads the line beginning at `pos`, without its line feed.
551///
552/// A final line with no line feed is a truncated stream, not a line, so it is
553/// refused rather than guessed at.
554fn read_line(src: &[u8], pos: usize, eof: bool)
555 -> Outcome<Line<'_>>
556{
557 if pos >= src.len() {
558 return Ok(if eof { Line::End } else { Line::Need });
559 }
560 match src[pos..].iter().position(|b| *b == b'\n') {
561 Some(i) => Ok(Line::Got(&src[pos..pos + i], pos + i + 1)),
562 None => {
563 if eof {
564 Err(err!(
565 "The stream ends with an unterminated line \"{}\".",
566 show(&src[pos..]);
567 Decode, Input, Missing))
568 } else {
569 Ok(Line::Need)
570 }
571 },
572 }
573}
574
575fn first_line(src: &[u8]) -> &[u8] {
576 match src.iter().position(|b| *b == b'\n') {
577 Some(i) => &src[..i],
578 None => src,
579 }
580}
581
582/// Renders stream bytes for an error message, lossily and shortened.
583pub(crate) fn show(bytes: &[u8]) -> String {
584 let cut = bytes.len().min(SHOW_LIMIT);
585 let mut out = String::new();
586 for b in &bytes[..cut] {
587 match *b {
588 b'\n' => out.push_str("\\n"),
589 b'\t' => out.push_str("\\t"),
590 0x20..=0x7e => out.push(*b as char),
591 other => out.push_str(&fmt!("\\{:03o}", other)),
592 }
593 }
594 if bytes.len() > cut {
595 out.push_str("...");
596 }
597 out
598}
599
600
601// ---------------------------------------------------------------------------
602// Command parsing.
603// ---------------------------------------------------------------------------
604
605/// Parses one command from the front of `src`, giving the event and the bytes
606/// it occupied. `None` where the command is not yet complete.
607fn parse_command(src: &[u8], eof: bool)
608 -> Outcome<Option<(Event, usize)>>
609{
610 let (line, next) = match res!(read_line(src, 0, eof)) {
611 Line::Got(line, next) => (line, next),
612 Line::End => return Ok(None),
613 Line::Need => return Ok(None),
614 };
615 if line == b"blob" {
616 parse_blob(src, next, eof)
617 } else if let Some(rest) = after(line, b"commit ") {
618 parse_commit(src, next, rest, eof)
619 } else if let Some(rest) = after(line, b"tag ") {
620 parse_tag(src, next, rest, eof)
621 } else if let Some(rest) = after(line, b"reset ") {
622 parse_reset(src, next, rest, eof)
623 } else if line == b"alias" {
624 parse_alias(src, next, eof)
625 } else if let Some(rest) = after(line, b"progress ") {
626 Ok(Some((Event::Progress(rest.to_vec()), next)))
627 } else if line == b"progress" {
628 Ok(Some((Event::Progress(Vec::new()), next)))
629 } else if line == b"checkpoint" {
630 Ok(Some((Event::Checkpoint, next)))
631 } else if line == b"done" {
632 Ok(Some((Event::Done, next)))
633 } else if let Some(rest) = after(line, b"feature ") {
634 let event = res!(parse_feature(rest));
635 Ok(Some((event, next)))
636 } else if let Some(rest) = after(line, b"option ") {
637 Ok(Some((Event::Opt(rest.to_vec()), next)))
638 } else if line.starts_with(b"ls ")
639 || line.starts_with(b"cat-blob ")
640 || line.starts_with(b"get-mark ")
641 {
642 Err(err!(
643 "Command line \"{}\" asks the consumer a question. Such commands \
644 exist for a frontend driving fast-import interactively; a stream \
645 from git fast-export never contains one, and this parser answers \
646 nothing.", show(line);
647 Decode, Input, NoImpl))
648 } else {
649 Err(err!(
650 "Command line \"{}\" is not a fast-import command this parser \
651 knows.", show(line);
652 Decode, Input, Invalid))
653 }
654}
655
656fn after<'a>(line: &'a [u8], prefix: &[u8]) -> Option<&'a [u8]> {
657 if line.starts_with(prefix) {
658 Some(&line[prefix.len()..])
659 } else {
660 None
661 }
662}
663
664/// The `blob` line has already been consumed.
665fn parse_blob(src: &[u8], mut p: usize, eof: bool)
666 -> Outcome<Option<(Event, usize)>>
667{
668 let mut mark = None;
669 let mut original_oid = None;
670 loop {
671 let (line, next) = match res!(read_line(src, p, eof)) {
672 Line::Got(line, next) => (line, next),
673 Line::End => return Err(err!(
674 "A blob command ends before its data.";
675 Decode, Input, Missing)),
676 Line::Need => return Ok(None),
677 };
678 if let Some(rest) = after(line, b"mark ") {
679 mark = Some(res!(parse_mark(rest)));
680 p = next;
681 } else if let Some(rest) = after(line, b"original-oid ") {
682 original_oid = Some(res!(parse_ascii(rest, "original-oid")));
683 p = next;
684 } else if line.starts_with(b"data") {
685 break;
686 } else {
687 return Err(err!(
688 "Line \"{}\" belongs to a blob command, where only mark, \
689 original-oid and data are allowed.", show(line);
690 Decode, Input, Invalid));
691 }
692 }
693 let (data, p) = match res!(take_data(src, p, eof)) {
694 Some(pair) => pair,
695 None => return Ok(None),
696 };
697 Ok(Some((Event::Blob(Blob { mark, original_oid, data }), p)))
698}
699
700/// The `commit` line has already been consumed.
701fn parse_commit(src: &[u8], mut p: usize, refname: &[u8], eof: bool)
702 -> Outcome<Option<(Event, usize)>>
703{
704 let refname = res!(parse_ascii(refname, "commit reference"));
705 let mut mark = None;
706 let mut original_oid = None;
707 let mut author = None;
708 let mut committer = None;
709 let mut encoding = None;
710 let mut gpgsig = None;
711 // The header runs to the message, which is the one part every commit has.
712 loop {
713 let (line, next) = match res!(read_line(src, p, eof)) {
714 Line::Got(line, next) => (line, next),
715 Line::End => return Err(err!(
716 "Commit on {} ends before its message.", refname;
717 Decode, Input, Missing)),
718 Line::Need => return Ok(None),
719 };
720 if let Some(rest) = after(line, b"mark ") {
721 mark = Some(res!(parse_mark(rest)));
722 p = next;
723 } else if let Some(rest) = after(line, b"original-oid ") {
724 original_oid = Some(res!(parse_ascii(rest, "original-oid")));
725 p = next;
726 } else if let Some(rest) = after(line, b"author ") {
727 author = Some(res!(parse_person(rest, "author")));
728 p = next;
729 } else if let Some(rest) = after(line, b"committer ") {
730 committer = Some(res!(parse_person(rest, "committer")));
731 p = next;
732 } else if let Some(rest) = after(line, b"encoding ") {
733 encoding = Some(rest.to_vec());
734 p = next;
735 } else if let Some(rest) = after(line, b"gpgsig ") {
736 let (algo, format) = res!(parse_gpgsig_head(rest));
737 let (sig, after_sig) = match res!(take_data(src, next, eof)) {
738 Some(pair) => pair,
739 None => return Ok(None),
740 };
741 gpgsig = Some(GpgSig { algo, format, sig });
742 p = after_sig;
743 } else if line.starts_with(b"data") {
744 break;
745 } else {
746 return Err(err!(
747 "Line \"{}\" appears in the header of the commit on {}, where \
748 it is not a header this parser knows.", show(line), refname;
749 Decode, Input, Invalid));
750 }
751 }
752 let committer = match committer {
753 Some(person) => person,
754 None => return Err(err!(
755 "Commit on {} has no committer line, which the format requires.",
756 refname;
757 Decode, Input, Missing)),
758 };
759 let (message, mut p) = match res!(take_data(src, p, eof)) {
760 Some(pair) => pair,
761 None => return Ok(None),
762 };
763
764 // Parents.
765 let mut from = None;
766 let mut merges = Vec::new();
767 loop {
768 let (line, next) = match res!(read_line(src, p, eof)) {
769 Line::Got(line, next) => (line, next),
770 Line::End => break,
771 Line::Need => return Ok(None),
772 };
773 if let Some(rest) = after(line, b"from ") {
774 if from.is_some() {
775 return Err(err!(
776 "Commit on {} carries a second from line, \"{}\".",
777 refname, show(line);
778 Decode, Input, Duplicate));
779 }
780 from = Some(res!(parse_objref(rest, "from")));
781 p = next;
782 } else if let Some(rest) = after(line, b"merge ") {
783 merges.push(res!(parse_objref(rest, "merge")));
784 p = next;
785 } else {
786 break;
787 }
788 }
789
790 // Changes to the tree.
791 let mut changes = Vec::new();
792 loop {
793 let (line, next) = match res!(read_line(src, p, eof)) {
794 Line::Got(line, next) => (line, next),
795 Line::End => break,
796 Line::Need => return Ok(None),
797 };
798 match res!(parse_file_change(src, line, next, eof)) {
799 Some(Some((change, after_change))) => {
800 changes.push(change);
801 p = after_change;
802 },
803 // Not a file command: the commit ends here, and the line stays.
804 Some(None) => break,
805 // A file command whose inline data has not all arrived.
806 None => return Ok(None),
807 }
808 }
809
810 Ok(Some((Event::Commit(Commit {
811 refname,
812 mark,
813 original_oid,
814 author,
815 committer,
816 encoding,
817 gpgsig,
818 message,
819 from,
820 merges,
821 changes,
822 }), p)))
823}
824
825/// The outer `Option` is `None` when more input is needed. The inner `Option`
826/// is `None` when the line is not a file command at all, which is how a commit
827/// ends.
828fn parse_file_change<'a>(src: &'a [u8], line: &'a [u8], next: usize, eof: bool)
829 -> Outcome<Option<Option<(FileChange, usize)>>>
830{
831 if let Some(rest) = after(line, b"M ") {
832 let (mode_bytes, at) = res!(take_token(rest, 0, "file mode"));
833 let mode = res!(FileMode::from_bytes(mode_bytes));
834 let (dataref, at) = res!(take_token(rest, at, "file data reference"));
835 let (path, _) = res!(take_path(rest, at, true));
836 if dataref == b"inline" {
837 let (bytes, after_data) = match res!(take_data(src, next, eof)) {
838 Some(pair) => pair,
839 None => return Ok(None),
840 };
841 Ok(Some(Some((
842 FileChange::Modify { mode, data: BlobRef::Inline(bytes), path },
843 after_data,
844 ))))
845 } else {
846 let data = res!(parse_blobref(dataref));
847 Ok(Some(Some((FileChange::Modify { mode, data, path }, next))))
848 }
849 } else if let Some(rest) = after(line, b"D ") {
850 let (path, _) = res!(take_path(rest, 0, true));
851 Ok(Some(Some((FileChange::Delete { path }, next))))
852 } else if let Some(rest) = after(line, b"C ") {
853 let (src_path, at) = res!(take_path(rest, 0, false));
854 let (dst_path, _) = res!(take_path(rest, at, true));
855 Ok(Some(Some((FileChange::Copy { src: src_path, dst: dst_path }, next))))
856 } else if let Some(rest) = after(line, b"R ") {
857 let (src_path, at) = res!(take_path(rest, 0, false));
858 let (dst_path, _) = res!(take_path(rest, at, true));
859 Ok(Some(Some((FileChange::Rename { src: src_path, dst: dst_path }, next))))
860 } else if line == b"deleteall" {
861 Ok(Some(Some((FileChange::DeleteAll, next))))
862 } else if let Some(rest) = after(line, b"N ") {
863 let (dataref, at) = res!(take_token(rest, 0, "note data reference"));
864 let commit = res!(parse_objref(&rest[at..], "note commit"));
865 if dataref == b"inline" {
866 let (bytes, after_data) = match res!(take_data(src, next, eof)) {
867 Some(pair) => pair,
868 None => return Ok(None),
869 };
870 Ok(Some(Some((
871 FileChange::Note { data: BlobRef::Inline(bytes), commit },
872 after_data,
873 ))))
874 } else {
875 let data = res!(parse_blobref(dataref));
876 Ok(Some(Some((FileChange::Note { data, commit }, next))))
877 }
878 } else {
879 Ok(Some(None))
880 }
881}
882
883/// The `tag` line has already been consumed.
884fn parse_tag(src: &[u8], mut p: usize, name: &[u8], eof: bool)
885 -> Outcome<Option<(Event, usize)>>
886{
887 let name = res!(parse_ascii(name, "tag name"));
888 let mut mark = None;
889 let mut original_oid = None;
890 let mut from = None;
891 let mut tagger = None;
892 loop {
893 let (line, next) = match res!(read_line(src, p, eof)) {
894 Line::Got(line, next) => (line, next),
895 Line::End => return Err(err!(
896 "Tag {} ends before its message.", name;
897 Decode, Input, Missing)),
898 Line::Need => return Ok(None),
899 };
900 if let Some(rest) = after(line, b"mark ") {
901 mark = Some(res!(parse_mark(rest)));
902 p = next;
903 } else if let Some(rest) = after(line, b"original-oid ") {
904 original_oid = Some(res!(parse_ascii(rest, "original-oid")));
905 p = next;
906 } else if let Some(rest) = after(line, b"from ") {
907 from = Some(res!(parse_objref(rest, "from")));
908 p = next;
909 } else if let Some(rest) = after(line, b"tagger ") {
910 tagger = Some(res!(parse_person(rest, "tagger")));
911 p = next;
912 } else if line.starts_with(b"data") {
913 break;
914 } else {
915 return Err(err!(
916 "Line \"{}\" appears in tag {}, where it is not a header this \
917 parser knows.", show(line), name;
918 Decode, Input, Invalid));
919 }
920 }
921 let from = match from {
922 Some(objref) => objref,
923 None => return Err(err!(
924 "Tag {} has no from line, so it names nothing.", name;
925 Decode, Input, Missing)),
926 };
927 let (message, p) = match res!(take_data(src, p, eof)) {
928 Some(pair) => pair,
929 None => return Ok(None),
930 };
931 Ok(Some((Event::Tag(Tag {
932 name,
933 mark,
934 original_oid,
935 from,
936 tagger,
937 message,
938 }), p)))
939}
940
941/// The `reset` line has already been consumed.
942fn parse_reset(src: &[u8], p: usize, refname: &[u8], eof: bool)
943 -> Outcome<Option<(Event, usize)>>
944{
945 let refname = res!(parse_ascii(refname, "reset reference"));
946 match res!(read_line(src, p, eof)) {
947 Line::Got(line, next) => {
948 match after(line, b"from ") {
949 Some(rest) => {
950 let from = Some(res!(parse_objref(rest, "from")));
951 Ok(Some((Event::Reset { refname, from }, next)))
952 },
953 None => Ok(Some((Event::Reset { refname, from: None }, p))),
954 }
955 },
956 Line::End => Ok(Some((Event::Reset { refname, from: None }, p))),
957 Line::Need => Ok(None),
958 }
959}
960
961/// The `alias` line has already been consumed.
962fn parse_alias(src: &[u8], p: usize, eof: bool)
963 -> Outcome<Option<(Event, usize)>>
964{
965 let (mark_line, p) = match res!(read_line(src, p, eof)) {
966 Line::Got(line, next) => (line, next),
967 Line::End => return Err(err!(
968 "An alias command ends before its mark.";
969 Decode, Input, Missing)),
970 Line::Need => return Ok(None),
971 };
972 let mark = match after(mark_line, b"mark ") {
973 Some(rest) => res!(parse_mark(rest)),
974 None => return Err(err!(
975 "An alias command must be followed by a mark line, not \"{}\".",
976 show(mark_line);
977 Decode, Input, Invalid)),
978 };
979 let (to_line, p) = match res!(read_line(src, p, eof)) {
980 Line::Got(line, next) => (line, next),
981 Line::End => return Err(err!(
982 "Alias of mark :{} ends before the object it names.", mark;
983 Decode, Input, Missing)),
984 Line::Need => return Ok(None),
985 };
986 let to = match after(to_line, b"to ") {
987 Some(rest) => res!(parse_objref(rest, "alias target")),
988 None => return Err(err!(
989 "An alias mark must be followed by a to line, not \"{}\".",
990 show(to_line);
991 Decode, Input, Invalid)),
992 };
993 Ok(Some((Event::Alias { mark, to }, p)))
994}
995
996fn parse_feature(rest: &[u8])
997 -> Outcome<Event>
998{
999 let (name, arg) = match rest.iter().position(|b| *b == b'=') {
1000 Some(i) => (
1001 res!(parse_ascii(&rest[..i], "feature name")),
1002 Some(res!(parse_ascii(&rest[i + 1..], "feature argument"))),
1003 ),
1004 None => (res!(parse_ascii(rest, "feature name")), None),
1005 };
1006 // Every date this parser reads is the raw format, so a stream declaring
1007 // another one has to be refused rather than silently misread.
1008 if name == "date-format" {
1009 match arg.as_deref() {
1010 Some("raw") | Some("raw-permissive") => {},
1011 other => return Err(err!(
1012 "The stream declares date-format {}, but only the raw format \
1013 is understood here.", other.unwrap_or("with no argument");
1014 Decode, Input, NoImpl)),
1015 }
1016 }
1017 Ok(Event::Feature { name, arg })
1018}
1019
1020
1021// ---------------------------------------------------------------------------
1022// Field parsing.
1023// ---------------------------------------------------------------------------
1024
1025/// Reads a `data` payload, in either the counted or the delimited form.
1026fn take_data(src: &[u8], p: usize, eof: bool)
1027 -> Outcome<Option<(Vec<u8>, usize)>>
1028{
1029 let (line, p) = match res!(read_line(src, p, eof)) {
1030 Line::Got(line, next) => (line, next),
1031 Line::End => return Err(err!(
1032 "The stream ends where a data command was expected.";
1033 Decode, Input, Missing)),
1034 Line::Need => return Ok(None),
1035 };
1036 let rest = match after(line, b"data ") {
1037 Some(rest) => rest,
1038 None => return Err(err!(
1039 "Line \"{}\" was expected to be a data command.", show(line);
1040 Decode, Input, Invalid)),
1041 };
1042 if let Some(delim) = after(rest, b"<<") {
1043 take_delimited_data(src, p, delim, eof)
1044 } else {
1045 let text = res!(parse_ascii(rest, "data byte count"));
1046 let count: usize = match text.parse() {
1047 Ok(n) => n,
1048 Err(_) => return Err(err!(
1049 "Data byte count \"{}\" is not a number.", text;
1050 Decode, Input, Invalid)),
1051 };
1052 if src.len() - p < count {
1053 return Ok(None);
1054 }
1055 let payload = src[p..p + count].to_vec();
1056 match skip_optional_lf(src, p + count, eof) {
1057 Some(end) => Ok(Some((payload, end))),
1058 None => Ok(None),
1059 }
1060 }
1061}
1062
1063/// Reads a delimited `data` payload, the terminator being a line holding
1064/// `delim` and nothing else.
1065fn take_delimited_data(src: &[u8], p: usize, delim: &[u8], eof: bool)
1066 -> Outcome<Option<(Vec<u8>, usize)>>
1067{
1068 if delim.is_empty() {
1069 return Err(err!(
1070 "A delimited data command gives an empty delimiter, which no line \
1071 can be distinguished by.";
1072 Decode, Input, Invalid));
1073 }
1074 let mut at = p;
1075 loop {
1076 let (line, next) = match res!(read_line(src, at, eof)) {
1077 Line::Got(line, next) => (line, next),
1078 Line::End => return Err(err!(
1079 "A delimited data command is never closed by its delimiter \
1080 \"{}\".", show(delim);
1081 Decode, Input, Missing)),
1082 Line::Need => return Ok(None),
1083 };
1084 if line == delim {
1085 // The line feed before the delimiter closes the payload and is not
1086 // part of it.
1087 let mut end = at;
1088 if end > p {
1089 end -= 1;
1090 }
1091 let payload = src[p..end].to_vec();
1092 return match skip_optional_lf(src, next, eof) {
1093 Some(after_lf) => Ok(Some((payload, after_lf))),
1094 None => Ok(None),
1095 };
1096 }
1097 at = next;
1098 }
1099}
1100
1101/// Steps over the line feed that may follow a payload, returning the offset
1102/// after it.
1103///
1104/// The format allows one, and git writes one, but it is not part of the
1105/// payload and it is not required. `None` asks for the one byte needed to tell
1106/// whether it is there, which is why a payload at the very end of a chunk waits
1107/// for the next chunk.
1108fn skip_optional_lf(src: &[u8], p: usize, eof: bool) -> Option<usize> {
1109 match src.get(p) {
1110 Some(b'\n') => Some(p + 1),
1111 Some(_) => Some(p),
1112 None => if eof { Some(p) } else { None },
1113 }
1114}
1115
1116/// Parses a mark line's argument, `:N`.
1117fn parse_mark(rest: &[u8])
1118 -> Outcome<u64>
1119{
1120 match after(rest, b":") {
1121 Some(digits) => {
1122 let text = res!(parse_ascii(digits, "mark number"));
1123 match text.parse() {
1124 Ok(n) => Ok(n),
1125 Err(_) => Err(err!(
1126 "Mark number \"{}\" is not a number.", text;
1127 Decode, Input, Invalid)),
1128 }
1129 },
1130 None => Err(err!(
1131 "Mark \"{}\" does not begin with a colon.", show(rest);
1132 Decode, Input, Invalid)),
1133 }
1134}
1135
1136/// Parses an object reference, which is either a mark or a name.
1137fn parse_objref(rest: &[u8], what: &str)
1138 -> Outcome<ObjRef>
1139{
1140 if rest.is_empty() {
1141 return Err(err!(
1142 "The {} names nothing.", what;
1143 Decode, Input, Missing));
1144 }
1145 if rest[0] == b':' {
1146 Ok(ObjRef::Mark(res!(parse_mark(rest))))
1147 } else {
1148 Ok(ObjRef::Name(res!(parse_ascii(rest, what))))
1149 }
1150}
1151
1152/// Parses a file data reference, which is either a mark or a name. The inline
1153/// form is handled by the caller, which has to read the payload that follows.
1154fn parse_blobref(token: &[u8])
1155 -> Outcome<BlobRef>
1156{
1157 if token.is_empty() {
1158 return Err(err!(
1159 "A file command names no content.";
1160 Decode, Input, Missing));
1161 }
1162 if token[0] == b':' {
1163 Ok(BlobRef::Mark(res!(parse_mark(token))))
1164 } else {
1165 Ok(BlobRef::Name(res!(parse_ascii(token, "file data reference"))))
1166 }
1167}
1168
1169/// Parses the `<algo> <format>` that introduces a commit signature.
1170fn parse_gpgsig_head(rest: &[u8])
1171 -> Outcome<(String, String)>
1172{
1173 match rest.iter().position(|b| *b == b' ') {
1174 Some(i) => Ok((
1175 res!(parse_ascii(&rest[..i], "signature hash algorithm")),
1176 res!(parse_ascii(&rest[i + 1..], "signature format")),
1177 )),
1178 None => Err(err!(
1179 "A gpgsig line must give a hash algorithm and a format, not \
1180 \"{}\".", show(rest);
1181 Decode, Input, Invalid)),
1182 }
1183}
1184
1185/// Parses an identity line: an optional name, an email in angle brackets, and
1186/// a time.
1187fn parse_person(rest: &[u8], what: &str)
1188 -> Outcome<Person>
1189{
1190 let open = match rest.iter().position(|b| *b == b'<') {
1191 Some(i) => i,
1192 None => return Err(err!(
1193 "The {} line \"{}\" has no email in angle brackets.",
1194 what, show(rest);
1195 Decode, Input, Invalid)),
1196 };
1197 let close = match rest[open..].iter().position(|b| *b == b'>') {
1198 Some(i) => open + i,
1199 None => return Err(err!(
1200 "The {} line \"{}\" opens an email but never closes it.",
1201 what, show(rest);
1202 Decode, Input, Invalid)),
1203 };
1204 // One space separates the name from the email, and belongs to neither.
1205 let mut name_end = open;
1206 if name_end > 0 && rest[name_end - 1] == b' ' {
1207 name_end -= 1;
1208 }
1209 let name = rest[..name_end].to_vec();
1210 let email = rest[open + 1..close].to_vec();
1211 let tail = &rest[close + 1..];
1212 let tail = match after(tail, b" ") {
1213 Some(t) => t,
1214 None => return Err(err!(
1215 "The {} line \"{}\" has no time after its email.", what, show(rest);
1216 Decode, Input, Missing)),
1217 };
1218 let when = res!(parse_when(tail, what));
1219 Ok(Person { name, email, when })
1220}
1221
1222/// Parses the raw date format: seconds since the epoch, then an offset.
1223fn parse_when(tail: &[u8], what: &str)
1224 -> Outcome<When>
1225{
1226 let sp = match tail.iter().position(|b| *b == b' ') {
1227 Some(i) => i,
1228 None => return Err(err!(
1229 "The {} time \"{}\" gives no timezone offset. Only git's raw date \
1230 format is read here.", what, show(tail);
1231 Decode, Input, Invalid)),
1232 };
1233 let secs_text = res!(parse_ascii(&tail[..sp], "timestamp"));
1234 let secs: i64 = match secs_text.parse() {
1235 Ok(n) => n,
1236 Err(_) => return Err(err!(
1237 "The {} timestamp \"{}\" is not a number of seconds.",
1238 what, secs_text;
1239 Decode, Input, Invalid)),
1240 };
1241 let tz = res!(parse_tz(&tail[sp + 1..], what));
1242 Ok(When { secs, tz })
1243}
1244
1245/// Parses a timezone offset written as a sign and four digits.
1246fn parse_tz(bytes: &[u8], what: &str)
1247 -> Outcome<TzOffset>
1248{
1249 let bad = || err!(
1250 "The {} timezone offset \"{}\" is not a sign followed by four digits.",
1251 what, show(bytes);
1252 Decode, Input, Invalid);
1253 if bytes.len() != 5 {
1254 return Err(bad());
1255 }
1256 let neg = match bytes[0] {
1257 b'+' => false,
1258 b'-' => true,
1259 _ => return Err(bad()),
1260 };
1261 if !bytes[1..].iter().all(|b| b.is_ascii_digit()) {
1262 return Err(bad());
1263 }
1264 let hours = i32::from(bytes[1] - b'0') * 10 + i32::from(bytes[2] - b'0');
1265 let mins = i32::from(bytes[3] - b'0') * 10 + i32::from(bytes[4] - b'0');
1266 let total = hours * 60 + mins;
1267 Ok(TzOffset { mins: if neg { -total } else { total }, neg })
1268}
1269
1270fn parse_ascii(bytes: &[u8], what: &str)
1271 -> Outcome<String>
1272{
1273 match std::str::from_utf8(bytes) {
1274 Ok(s) => Ok(s.to_string()),
1275 Err(_) => Err(err!(
1276 "The {} \"{}\" is not valid UTF-8.", what, show(bytes);
1277 Decode, Input, Invalid)),
1278 }
1279}
1280
1281/// Takes the next space separated token, returning it and the offset just past
1282/// the separating space.
1283fn take_token<'a>(line: &'a [u8], pos: usize, what: &str)
1284 -> Outcome<(&'a [u8], usize)>
1285{
1286 match line[pos..].iter().position(|b| *b == b' ') {
1287 Some(i) => Ok((&line[pos..pos + i], pos + i + 1)),
1288 None => Err(err!(
1289 "Line \"{}\" ends where the {} and what follows it were expected.",
1290 show(line), what;
1291 Decode, Input, Missing)),
1292 }
1293}
1294
1295
1296// ---------------------------------------------------------------------------
1297// Path quoting.
1298// ---------------------------------------------------------------------------
1299
1300/// Reads one path from `line`, unquoting it if it is quoted.
1301///
1302/// An unquoted path runs to the next space, or to the end of the line when it
1303/// is the last on the line. A quoted path runs to its closing quote whatever
1304/// it contains.
1305fn take_path(line: &[u8], pos: usize, last: bool)
1306 -> Outcome<(Vec<u8>, usize)>
1307{
1308 if pos > line.len() {
1309 return Err(err!(
1310 "Line \"{}\" ends where a path was expected.", show(line);
1311 Decode, Input, Missing));
1312 }
1313 if line.get(pos) == Some(&b'"') {
1314 return unquote_path(line, pos);
1315 }
1316 match line[pos..].iter().position(|b| *b == b' ') {
1317 Some(i) if !last => Ok((line[pos..pos + i].to_vec(), pos + i + 1)),
1318 _ if last => Ok((line[pos..].to_vec(), line.len())),
1319 _ => Err(err!(
1320 "Line \"{}\" gives one path where two were expected. An unquoted \
1321 path cannot contain a space, so a path that does must be quoted.",
1322 show(line);
1323 Decode, Input, Missing)),
1324 }
1325}
1326
1327/// Unquotes a C-style quoted path beginning at `pos`, which is its opening
1328/// quote.
1329fn unquote_path(line: &[u8], pos: usize)
1330 -> Outcome<(Vec<u8>, usize)>
1331{
1332 let mut out = Vec::new();
1333 let mut i = pos + 1;
1334 while i < line.len() {
1335 match line[i] {
1336 b'"' => {
1337 // A space after the closing quote separates this path from the
1338 // next, and belongs to neither.
1339 let mut next = i + 1;
1340 if line.get(next) == Some(&b' ') {
1341 next += 1;
1342 }
1343 return Ok((out, next));
1344 },
1345 b'\\' => {
1346 if i + 1 >= line.len() {
1347 return Err(err!(
1348 "Quoted path \"{}\" ends in a backslash.", show(line);
1349 Decode, Input, Invalid));
1350 }
1351 let esc = line[i + 1];
1352 i += 2;
1353 match esc {
1354 b'a' => out.push(0x07),
1355 b'b' => out.push(0x08),
1356 b'f' => out.push(0x0c),
1357 b'n' => out.push(b'\n'),
1358 b'r' => out.push(b'\r'),
1359 b't' => out.push(b'\t'),
1360 b'v' => out.push(0x0b),
1361 b'"' => out.push(b'"'),
1362 b'\\' => out.push(b'\\'),
1363 b'0'..=b'7' => {
1364 // Up to three octal digits, the first already read.
1365 let mut value = u32::from(esc - b'0');
1366 let mut digits = 1;
1367 while digits < 3 {
1368 match line.get(i) {
1369 Some(d @ b'0'..=b'7') => {
1370 value = value * 8 + u32::from(*d - b'0');
1371 i += 1;
1372 digits += 1;
1373 },
1374 _ => break,
1375 }
1376 }
1377 if value > 0xff {
1378 return Err(err!(
1379 "Quoted path \"{}\" contains the octal escape \
1380 \\{:o}, which is not a byte.", show(line), value;
1381 Decode, Input, Range));
1382 }
1383 out.push(value as u8);
1384 },
1385 other => return Err(err!(
1386 "Quoted path \"{}\" contains the escape \\{}, which \
1387 the format does not define.",
1388 show(line), show(&[other]);
1389 Decode, Input, Invalid)),
1390 }
1391 },
1392 byte => {
1393 out.push(byte);
1394 i += 1;
1395 },
1396 }
1397 }
1398 Err(err!(
1399 "Quoted path \"{}\" is never closed.", show(line);
1400 Decode, Input, Missing))
1401}
1402
1403
1404#[cfg(test)]
1405mod test {
1406 use super::*;
1407
1408 /// This is where a boundary bug shows: a parser that peeks past what it has
1409 /// been given behaves differently when the bytes trickle in.
1410 fn parse_both_ways(stream: &[u8])
1411 -> Outcome<Vec<Event>>
1412 {
1413 let whole = res!(parse_all(stream));
1414
1415 let mut parser = Parser::new();
1416 let mut trickled = Vec::new();
1417 for byte in stream {
1418 parser.feed(&[*byte]);
1419 loop {
1420 match res!(parser.next_event()) {
1421 Some(event) => trickled.push(event),
1422 None => break,
1423 }
1424 }
1425 }
1426 parser.end();
1427 while let Some(event) = res!(parser.next_event()) {
1428 trickled.push(event);
1429 }
1430
1431 // And once more in chunks of seven, an unhelpful size on purpose.
1432 let mut parser = Parser::new();
1433 let mut chunked = Vec::new();
1434 for chunk in stream.chunks(7) {
1435 parser.feed(chunk);
1436 loop {
1437 match res!(parser.next_event()) {
1438 Some(event) => chunked.push(event),
1439 None => break,
1440 }
1441 }
1442 }
1443 parser.end();
1444 while let Some(event) = res!(parser.next_event()) {
1445 chunked.push(event);
1446 }
1447
1448 assert_eq!(whole, trickled, "byte at a time differed from all at once");
1449 assert_eq!(whole, chunked, "seven byte chunks differed from all at once");
1450 Ok(whole)
1451 }
1452
1453 fn ada() -> Person {
1454 Person {
1455 name: b"Ada Lovelace".to_vec(),
1456 email: b"ada@example.org".to_vec(),
1457 when: When { secs: 1700000000, tz: TzOffset { mins: 600, neg: false } },
1458 }
1459 }
1460
1461 #[test]
1462 fn blob_with_mark() -> Outcome<()> {
1463 let events = res!(parse_both_ways(b"blob\nmark :1\ndata 5\nhello\n"));
1464 assert_eq!(events, vec![Event::Blob(Blob {
1465 mark: Some(1),
1466 original_oid: None,
1467 data: b"hello".to_vec(),
1468 })]);
1469 Ok(())
1470 }
1471
1472 /// A blob payload holding NUL bytes and bare line feeds survives whole, and
1473 /// is never mistaken for stream syntax.
1474 #[test]
1475 fn blob_payload_is_binary_safe() -> Outcome<()> {
1476 let payload: &[u8] = b"\x00\x01\ndata 99\nblob\n\x00commit refs/heads/x\n";
1477 let mut stream = Vec::new();
1478 stream.extend_from_slice(b"blob\nmark :7\noriginal-oid ");
1479 stream.extend_from_slice(b"1234567890abcdef1234567890abcdef12345678\n");
1480 stream.extend_from_slice(fmt!("data {}\n", payload.len()).as_bytes());
1481 stream.extend_from_slice(payload);
1482 stream.extend_from_slice(b"\n");
1483 let events = res!(parse_both_ways(&stream));
1484 assert_eq!(events, vec![Event::Blob(Blob {
1485 mark: Some(7),
1486 original_oid: Some(fmt!("1234567890abcdef1234567890abcdef12345678")),
1487 data: payload.to_vec(),
1488 })]);
1489 Ok(())
1490 }
1491
1492 #[test]
1493 fn empty_payload() -> Outcome<()> {
1494 let events = res!(parse_both_ways(b"blob\ndata 0\n"));
1495 assert_eq!(events, vec![Event::Blob(Blob {
1496 mark: None,
1497 original_oid: None,
1498 data: Vec::new(),
1499 })]);
1500 Ok(())
1501 }
1502
1503 /// A payload with no line feed of its own after it still parses, since the
1504 /// trailing line feed is optional.
1505 #[test]
1506 fn payload_without_trailing_lf() -> Outcome<()> {
1507 let events = res!(parse_both_ways(b"blob\ndata 2\nhi"));
1508 assert_eq!(events, vec![Event::Blob(Blob {
1509 mark: None,
1510 original_oid: None,
1511 data: b"hi".to_vec(),
1512 })]);
1513 Ok(())
1514 }
1515
1516 #[test]
1517 fn commit_with_changes() -> Outcome<()> {
1518 let stream: &[u8] = b"commit refs/heads/main\n\
1519 mark :3\n\
1520 author Ada Lovelace <ada@example.org> 1700000000 +1000\n\
1521 committer Ada Lovelace <ada@example.org> 1700000000 +1000\n\
1522 data 13\n\
1523 first commit\n\
1524 M 100644 :1 a.txt\n\
1525 M 100755 :2 run.sh\n\
1526 D gone.txt\n\
1527 \n";
1528 let events = res!(parse_both_ways(stream));
1529 assert_eq!(events, vec![Event::Commit(Commit {
1530 refname: fmt!("refs/heads/main"),
1531 mark: Some(3),
1532 original_oid: None,
1533 author: Some(ada()),
1534 committer: ada(),
1535 encoding: None,
1536 gpgsig: None,
1537 message: b"first commit\n".to_vec(),
1538 from: None,
1539 merges: Vec::new(),
1540 changes: vec![
1541 FileChange::Modify {
1542 mode: FileMode::Normal,
1543 data: BlobRef::Mark(1),
1544 path: b"a.txt".to_vec(),
1545 },
1546 FileChange::Modify {
1547 mode: FileMode::Executable,
1548 data: BlobRef::Mark(2),
1549 path: b"run.sh".to_vec(),
1550 },
1551 FileChange::Delete { path: b"gone.txt".to_vec() },
1552 ],
1553 })]);
1554 Ok(())
1555 }
1556
1557 /// A merge commit carries a first parent and every other parent in order.
1558 #[test]
1559 fn merge_commit_has_two_parents() -> Outcome<()> {
1560 let stream: &[u8] = b"commit refs/heads/main\n\
1561 mark :9\n\
1562 committer Ada Lovelace <ada@example.org> 1700000000 +1000\n\
1563 data 11\n\
1564 merge side\n\
1565 from :4\n\
1566 merge :6\n\
1567 merge 0123456789abcdef0123456789abcdef01234567\n\
1568 \n";
1569 let events = res!(parse_both_ways(stream));
1570 match &events[0] {
1571 Event::Commit(commit) => {
1572 assert_eq!(commit.from, Some(ObjRef::Mark(4)));
1573 assert_eq!(commit.merges, vec![
1574 ObjRef::Mark(6),
1575 ObjRef::Name(fmt!("0123456789abcdef0123456789abcdef01234567")),
1576 ]);
1577 assert!(commit.author.is_none());
1578 },
1579 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1580 }
1581 Ok(())
1582 }
1583
1584 /// A copy, a rename and a deleteall all parse, and the deleteall keeps its
1585 /// place in the order.
1586 #[test]
1587 fn copy_rename_and_deleteall() -> Outcome<()> {
1588 let stream: &[u8] = b"commit refs/heads/main\n\
1589 committer A <a@b> 0 +0000\n\
1590 data 0\n\
1591 deleteall\n\
1592 C src.txt copy.txt\n\
1593 R old.txt new.txt\n\
1594 \n";
1595 let events = res!(parse_both_ways(stream));
1596 match &events[0] {
1597 Event::Commit(commit) => assert_eq!(commit.changes, vec![
1598 FileChange::DeleteAll,
1599 FileChange::Copy {
1600 src: b"src.txt".to_vec(),
1601 dst: b"copy.txt".to_vec(),
1602 },
1603 FileChange::Rename {
1604 src: b"old.txt".to_vec(),
1605 dst: b"new.txt".to_vec(),
1606 },
1607 ]),
1608 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1609 }
1610 Ok(())
1611 }
1612
1613 #[test]
1614 fn inline_file_content() -> Outcome<()> {
1615 let stream: &[u8] = b"commit refs/heads/main\n\
1616 committer A <a@b> 0 +0000\n\
1617 data 0\n\
1618 M 100644 inline hello.txt\n\
1619 data 6\n\
1620 world\n\
1621 M 120000 inline link\n\
1622 data 7\n\
1623 target\n\
1624 \n";
1625 let events = res!(parse_both_ways(stream));
1626 match &events[0] {
1627 Event::Commit(commit) => assert_eq!(commit.changes, vec![
1628 FileChange::Modify {
1629 mode: FileMode::Normal,
1630 data: BlobRef::Inline(b"world\n".to_vec()),
1631 path: b"hello.txt".to_vec(),
1632 },
1633 FileChange::Modify {
1634 mode: FileMode::Symlink,
1635 data: BlobRef::Inline(b"target\n".to_vec()),
1636 path: b"link".to_vec(),
1637 },
1638 ]),
1639 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1640 }
1641 Ok(())
1642 }
1643
1644 #[test]
1645 fn quoted_paths_unquote() -> Outcome<()> {
1646 let stream: &[u8] = b"commit refs/heads/main\n\
1647 committer A <a@b> 0 +0000\n\
1648 data 0\n\
1649 M 100644 :1 \"has space.txt\"\n\
1650 M 100644 :1 \"new\\nline\"\n\
1651 M 100644 :1 \"tab\\there\"\n\
1652 M 100644 :1 \"back\\\\slash\"\n\
1653 M 100644 :1 \"quote\\\"mark\"\n\
1654 M 100644 :1 \"bell\\a\\b\\f\\r\\v\"\n\
1655 M 100644 :1 \"octal\\303\\251\"\n\
1656 M 100644 :1 \"short\\7octal\"\n\
1657 D \"deleted \\303\\266.txt\"\n\
1658 R \"old name.txt\" \"new name.txt\"\n\
1659 C \"a b\" plain.txt\n\
1660 \n";
1661 let events = res!(parse_both_ways(stream));
1662 let want: Vec<FileChange> = vec![
1663 b"has space.txt".to_vec(),
1664 b"new\nline".to_vec(),
1665 b"tab\there".to_vec(),
1666 b"back\\slash".to_vec(),
1667 b"quote\"mark".to_vec(),
1668 b"bell\x07\x08\x0c\r\x0b".to_vec(),
1669 b"octal\xc3\xa9".to_vec(),
1670 b"short\x07octal".to_vec(),
1671 ].into_iter().map(|path| FileChange::Modify {
1672 mode: FileMode::Normal,
1673 data: BlobRef::Mark(1),
1674 path,
1675 }).collect();
1676 match &events[0] {
1677 Event::Commit(commit) => {
1678 assert_eq!(&commit.changes[..8], &want[..]);
1679 assert_eq!(commit.changes[8], FileChange::Delete {
1680 path: b"deleted \xc3\xb6.txt".to_vec(),
1681 });
1682 assert_eq!(commit.changes[9], FileChange::Rename {
1683 src: b"old name.txt".to_vec(),
1684 dst: b"new name.txt".to_vec(),
1685 });
1686 assert_eq!(commit.changes[10], FileChange::Copy {
1687 src: b"a b".to_vec(),
1688 dst: b"plain.txt".to_vec(),
1689 });
1690 },
1691 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1692 }
1693 Ok(())
1694 }
1695
1696 /// A path that is last on its line runs to the end of the line, spaces and
1697 /// all. Git quotes such a path, but the format allows it unquoted and
1698 /// fast-import reads it, so this parser does too.
1699 #[test]
1700 fn unquoted_path_may_hold_spaces() -> Outcome<()> {
1701 let stream: &[u8] = b"commit refs/heads/main\n\
1702 committer A <a@b> 0 +0000\n\
1703 data 0\n\
1704 M 100644 :1 plain path.txt\n\
1705 D another path.txt\n\
1706 \n";
1707 let events = res!(parse_both_ways(stream));
1708 match &events[0] {
1709 Event::Commit(commit) => assert_eq!(commit.changes, vec![
1710 FileChange::Modify {
1711 mode: FileMode::Normal,
1712 data: BlobRef::Mark(1),
1713 path: b"plain path.txt".to_vec(),
1714 },
1715 FileChange::Delete { path: b"another path.txt".to_vec() },
1716 ]),
1717 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1718 }
1719 Ok(())
1720 }
1721
1722 #[test]
1723 fn unknown_escape_is_refused() -> Outcome<()> {
1724 let stream: &[u8] = b"commit refs/heads/main\n\
1725 committer A <a@b> 0 +0000\n\
1726 data 0\n\
1727 M 100644 :1 \"what\\qnow\"\n\
1728 \n";
1729 assert!(parse_all(stream).is_err());
1730 Ok(())
1731 }
1732
1733 /// A delimited payload runs to a line holding just the delimiter, and that
1734 /// line's own content may appear inside the payload without ending it.
1735 #[test]
1736 fn delimited_data() -> Outcome<()> {
1737 let stream: &[u8] = b"commit refs/heads/main\n\
1738 committer A <a@b> 0 +0000\n\
1739 data <<END\n\
1740 a message\n\
1741 with ENDish lines\n\
1742 END\n\
1743 \n";
1744 let events = res!(parse_both_ways(stream));
1745 match &events[0] {
1746 Event::Commit(commit) => assert_eq!(
1747 commit.message,
1748 b"a message\nwith ENDish lines".to_vec(),
1749 ),
1750 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1751 }
1752 Ok(())
1753 }
1754
1755 #[test]
1756 fn delimited_data_empty() -> Outcome<()> {
1757 let events = res!(parse_both_ways(b"blob\ndata <<EOF\nEOF\n"));
1758 assert_eq!(events, vec![Event::Blob(Blob {
1759 mark: None,
1760 original_oid: None,
1761 data: Vec::new(),
1762 })]);
1763 Ok(())
1764 }
1765
1766 #[test]
1767 fn tag_parses() -> Outcome<()> {
1768 let stream: &[u8] = b"tag v1\n\
1769 mark :12\n\
1770 from :7\n\
1771 tagger Ada Lovelace <ada@example.org> 1700000000 +1000\n\
1772 data 12\n\
1773 release one\n\
1774 \n";
1775 let events = res!(parse_both_ways(stream));
1776 assert_eq!(events, vec![Event::Tag(Tag {
1777 name: fmt!("v1"),
1778 mark: Some(12),
1779 original_oid: None,
1780 from: ObjRef::Mark(7),
1781 tagger: Some(ada()),
1782 message: b"release one\n".to_vec(),
1783 })]);
1784 Ok(())
1785 }
1786
1787 /// A reset parses with and without a from line, and one without does not
1788 /// swallow the command that follows.
1789 #[test]
1790 fn reset_with_and_without_from() -> Outcome<()> {
1791 let stream: &[u8] = b"reset refs/heads/side\n\
1792 from :3\n\
1793 reset refs/heads/other\n\
1794 blob\n\
1795 data 1\n\
1796 x\n";
1797 let events = res!(parse_both_ways(stream));
1798 assert_eq!(events.len(), 3);
1799 assert_eq!(events[0], Event::Reset {
1800 refname: fmt!("refs/heads/side"),
1801 from: Some(ObjRef::Mark(3)),
1802 });
1803 assert_eq!(events[1], Event::Reset {
1804 refname: fmt!("refs/heads/other"),
1805 from: None,
1806 });
1807 assert_eq!(events[2], Event::Blob(Blob {
1808 mark: None,
1809 original_oid: None,
1810 data: b"x".to_vec(),
1811 }));
1812 Ok(())
1813 }
1814
1815 /// The commands that carry no repository content each parse.
1816 #[test]
1817 fn housekeeping_commands() -> Outcome<()> {
1818 let stream: &[u8] = b"feature done\n\
1819 feature date-format=raw\n\
1820 option quiet\n\
1821 progress 3 of 9 objects\n\
1822 checkpoint\n\
1823 alias\n\
1824 mark :5\n\
1825 to :4\n\
1826 done\n";
1827 let events = res!(parse_both_ways(stream));
1828 assert_eq!(events, vec![
1829 Event::Feature { name: fmt!("done"), arg: None },
1830 Event::Feature { name: fmt!("date-format"), arg: Some(fmt!("raw")) },
1831 Event::Opt(b"quiet".to_vec()),
1832 Event::Progress(b"3 of 9 objects".to_vec()),
1833 Event::Checkpoint,
1834 Event::Alias { mark: 5, to: ObjRef::Mark(4) },
1835 Event::Done,
1836 ]);
1837 Ok(())
1838 }
1839
1840 /// A date format other than raw is refused, since reading it as raw would
1841 /// silently misdate every commit.
1842 #[test]
1843 fn foreign_date_format_is_refused() -> Outcome<()> {
1844 assert!(parse_all(b"feature date-format=rfc2822\n").is_err());
1845 Ok(())
1846 }
1847
1848 #[test]
1849 fn file_modes() -> Outcome<()> {
1850 let cases: [(&[u8], FileMode); 8] = [
1851 (b"100644", FileMode::Normal),
1852 (b"644", FileMode::Normal),
1853 (b"100755", FileMode::Executable),
1854 (b"755", FileMode::Executable),
1855 (b"120000", FileMode::Symlink),
1856 (b"160000", FileMode::Gitlink),
1857 (b"040000", FileMode::Subdirectory),
1858 (b"40000", FileMode::Subdirectory),
1859 ];
1860 for (bytes, want) in cases {
1861 assert_eq!(res!(FileMode::from_bytes(bytes)), want);
1862 }
1863 assert!(FileMode::from_bytes(b"100664").is_err());
1864 Ok(())
1865 }
1866
1867 /// Three of git's five modes name something the vocabulary can hold, and the
1868 /// other two name something that is not a file with bytes in it.
1869 #[test]
1870 fn modes_that_name_a_file() -> Outcome<()> {
1871 assert_eq!(FileMode::Normal.as_op_mode(), Some(Mode::Normal));
1872 assert_eq!(FileMode::Executable.as_op_mode(), Some(Mode::Executable));
1873 assert_eq!(FileMode::Symlink.as_op_mode(), Some(Mode::Symlink));
1874 assert_eq!(FileMode::Gitlink.as_op_mode(), None,
1875 "a submodule is another repository, not this one's file");
1876 assert_eq!(FileMode::Subdirectory.as_op_mode(), None,
1877 "a tree entry is not content");
1878 Ok(())
1879 }
1880
1881 /// An identity line with no name, and one with an offset west of UTC,
1882 /// both parse.
1883 #[test]
1884 fn identity_lines() -> Outcome<()> {
1885 let person = res!(parse_person(b"<nobody@example.org> 1700000200 -0530", "author"));
1886 assert_eq!(person.name, Vec::<u8>::new());
1887 assert_eq!(person.email, b"nobody@example.org".to_vec());
1888 assert_eq!(person.when.secs, 1700000200);
1889 assert_eq!(person.when.tz.mins, -330);
1890 assert_eq!(fmt!("{}", person.when), "1700000200 -0530");
1891 Ok(())
1892 }
1893
1894 /// The unknown timezone `-0000` keeps its sign, which is the only thing
1895 /// that distinguishes it from `+0000`.
1896 #[test]
1897 fn unknown_timezone_keeps_its_sign() -> Outcome<()> {
1898 let west = res!(parse_person(b"A <a@b> 0 -0000", "author"));
1899 let east = res!(parse_person(b"A <a@b> 0 +0000", "author"));
1900 assert_eq!(west.when.tz.mins, 0);
1901 assert_eq!(east.when.tz.mins, 0);
1902 assert_ne!(west.when.tz, east.when.tz);
1903 assert_eq!(fmt!("{}", west.when.tz), "-0000");
1904 assert_eq!(fmt!("{}", east.when.tz), "+0000");
1905 Ok(())
1906 }
1907
1908 #[test]
1909 fn non_utf8_name_survives() -> Outcome<()> {
1910 let mut line = Vec::new();
1911 line.extend_from_slice(b"Rene\xe9 <r@example.org> 0 +0000");
1912 let person = res!(parse_person(&line, "author"));
1913 assert_eq!(person.name, b"Rene\xe9".to_vec());
1914 assert_eq!(person.name_lossy(), "Rene\u{fffd}");
1915 Ok(())
1916 }
1917
1918 #[test]
1919 fn gpgsig_is_carried() -> Outcome<()> {
1920 let stream: &[u8] = b"commit refs/heads/main\n\
1921 committer A <a@b> 0 +0000\n\
1922 gpgsig sha1 openpgp\n\
1923 data 20\n\
1924 -----BEGIN PGP-----\n\
1925 data 2\n\
1926 m\n\
1927 \n";
1928 let events = res!(parse_both_ways(stream));
1929 match &events[0] {
1930 Event::Commit(commit) => {
1931 assert_eq!(commit.gpgsig, Some(GpgSig {
1932 algo: fmt!("sha1"),
1933 format: fmt!("openpgp"),
1934 sig: b"-----BEGIN PGP-----\n".to_vec(),
1935 }));
1936 assert_eq!(commit.message, b"m\n".to_vec());
1937 },
1938 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1939 }
1940 Ok(())
1941 }
1942
1943 #[test]
1944 fn notemodify_parses() -> Outcome<()> {
1945 let stream: &[u8] = b"commit refs/notes/commits\n\
1946 committer A <a@b> 0 +0000\n\
1947 data 0\n\
1948 N :2 :1\n\
1949 N inline :3\n\
1950 data 5\n\
1951 note\n\
1952 \n";
1953 let events = res!(parse_both_ways(stream));
1954 match &events[0] {
1955 Event::Commit(commit) => assert_eq!(commit.changes, vec![
1956 FileChange::Note {
1957 data: BlobRef::Mark(2),
1958 commit: ObjRef::Mark(1),
1959 },
1960 FileChange::Note {
1961 data: BlobRef::Inline(b"note\n".to_vec()),
1962 commit: ObjRef::Mark(3),
1963 },
1964 ]),
1965 other => return Err(err!("Expected a commit, got {:?}.", other; Test)),
1966 }
1967 Ok(())
1968 }
1969
1970 #[test]
1971 fn unknown_command_names_the_line() -> Outcome<()> {
1972 match parse_all(b"blob\ndata 0\nfrobnicate everything\n") {
1973 Ok(events) => Err(err!(
1974 "Expected an error, parsed {:?}.", events;
1975 Test)),
1976 Err(e) => {
1977 let text = fmt!("{}", e);
1978 assert!(
1979 text.contains("frobnicate everything"),
1980 "error did not name the line: {}", text,
1981 );
1982 Ok(())
1983 },
1984 }
1985 }
1986
1987 /// The commands a frontend uses to interrogate fast-import are refused by
1988 /// name, since there is nothing here to answer them.
1989 #[test]
1990 fn interrogation_commands_are_refused() -> Outcome<()> {
1991 for line in [
1992 &b"ls :1 path\n"[..],
1993 &b"cat-blob :1\n"[..],
1994 &b"get-mark :1\n"[..],
1995 ] {
1996 assert!(parse_all(line).is_err(), "accepted {:?}", line);
1997 }
1998 Ok(())
1999 }
2000
2001 /// A stream cut off part way through a payload is an error once the caller
2002 /// says no more is coming, not a short blob.
2003 #[test]
2004 fn truncated_stream_is_an_error() -> Outcome<()> {
2005 assert!(parse_all(b"blob\ndata 10\nshort").is_err());
2006 assert!(parse_all(b"blob\nmark :1\n").is_err());
2007 assert!(parse_all(b"commit refs/heads/main\ncommitter A <a@b> 0 +0000\n").is_err());
2008 Ok(())
2009 }
2010
2011 /// A commit with no committer is refused, since the format requires one.
2012 #[test]
2013 fn commit_without_committer_is_refused() -> Outcome<()> {
2014 assert!(parse_all(b"commit refs/heads/main\ndata 0\n").is_err());
2015 Ok(())
2016 }
2017
2018 /// Until the caller declares the end of the stream, an incomplete command
2019 /// asks for more input rather than failing.
2020 #[test]
2021 fn incomplete_command_waits() -> Outcome<()> {
2022 let mut parser = Parser::new();
2023 parser.feed(b"blob\ndata 5\nhel");
2024 assert_eq!(res!(parser.next_event()), None);
2025 assert!(!parser.is_exhausted());
2026 parser.feed(b"lo\n");
2027 assert!(res!(parser.next_event()).is_some());
2028 parser.end();
2029 assert_eq!(res!(parser.next_event()), None);
2030 assert!(parser.is_exhausted());
2031 Ok(())
2032 }
2033
2034 /// A whole stream of every construct parses the same however it is chunked.
2035 #[test]
2036 fn whole_stream_round_trip() -> Outcome<()> {
2037 let mut stream = Vec::new();
2038 stream.extend_from_slice(b"feature done\n");
2039 stream.extend_from_slice(b"blob\nmark :1\ndata 6\nhello\n\n");
2040 stream.extend_from_slice(b"blob\nmark :2\ndata 11\n");
2041 stream.extend_from_slice(b"\x00\x01\n\x02binary\x00");
2042 stream.extend_from_slice(b"\n");
2043 stream.extend_from_slice(b"reset refs/heads/side\n");
2044 stream.extend_from_slice(
2045 b"commit refs/heads/side\n\
2046 mark :3\n\
2047 author Ada Lovelace <ada@example.org> 1700000000 +1000\n\
2048 committer Ada Lovelace <ada@example.org> 1700000000 +1000\n\
2049 data 13\n\
2050 first commit\n\
2051 M 100644 :1 a.txt\n\
2052 M 100644 :2 \"bin \\303\\251.dat\"\n\
2053 \n");
2054 stream.extend_from_slice(
2055 b"commit refs/heads/main\n\
2056 mark :4\n\
2057 committer Ada Lovelace <ada@example.org> 1700000000 +1000\n\
2058 data <<EOM\n\
2059 rename a\n\
2060 EOM\n\
2061 from :3\n\
2062 R a.txt renamed.txt\n\
2063 \n");
2064 stream.extend_from_slice(
2065 b"tag v1\n\
2066 from :4\n\
2067 tagger Ada Lovelace <ada@example.org> 1700000000 +1000\n\
2068 data 12\n\
2069 release one\n\
2070 \n");
2071 stream.extend_from_slice(b"done\n");
2072 let events = res!(parse_both_ways(&stream));
2073 assert_eq!(events.len(), 8);
2074 match &events[6] {
2075 Event::Tag(tag) => assert_eq!(tag.name, "v1"),
2076 other => return Err(err!("Expected a tag, got {:?}.", other; Test)),
2077 }
2078 assert_eq!(events[7], Event::Done);
2079 Ok(())
2080 }
2081}