oxedyne/ore/cli/src/gitimport.rs
56.2 KiB, 196 runs
created by r2848102244:9, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Reading a git repository into an Ore one. |
| 2 | //! |
| 3 | //! Git keeps exactly one interface for foreign consumers of a repository, the |
| 4 | //! byte stream described by `git-fast-import(1)`, and this reads that. The |
| 5 | //! stream is obtained by running `git fast-export --all -M` in the repository |
| 6 | //! being imported and parsing its standard output as it arrives, through |
| 7 | //! `oxedyne_fe2o3_ore::fastexport`, which is a pure parser and spawns nothing |
| 8 | //! itself. |
| 9 | //! |
| 10 | //! # The translation |
| 11 | //! |
| 12 | //! - Each git author becomes a replica. A mapping already in `.ore/config` is |
| 13 | //! consulted first, so that one person writing under several identity lines -- |
| 14 | //! a name that changed, an address that lapsed -- imports as one author rather |
| 15 | //! than as several who never met. Otherwise the identifier is derived from the |
| 16 | //! identity line, so the same person imports as the same replica twice over, |
| 17 | //! and the derivation is written into the mapping. **Set the mapping before |
| 18 | //! importing if it is wanted at all**: a replica identifier is part of every |
| 19 | //! operation identifier its author writes, the header carrying it is covered by |
| 20 | //! the signature, and no later pass can correct it. |
| 21 | //! - Each commit becomes the operations its file changes call for, followed by |
| 22 | //! an `Op::Mark` carrying the commit itself: named by its subject, holding |
| 23 | //! the rest of the message as its body, dated by its author date, and |
| 24 | //! recording git's whole author line -- identity, seconds and zone offset -- |
| 25 | //! so that a mirror can write that back out unchanged. The mark is the |
| 26 | //! commit: a later commit takes the marks of its git parents as the parents |
| 27 | //! of its first operation, so the operation graph has the shape of the commit |
| 28 | //! graph. |
| 29 | //! - `M` becomes a create and a splice, or the splices that turn what the first |
| 30 | //! parent held into the blob; `D` becomes a delete; `R` becomes a rename, |
| 31 | //! since git says outright that one happened. |
| 32 | //! - A mode git records against a path becomes an `Op::FileMode` where it is not |
| 33 | //! the mode the file already has, so a script imports as a script and stays |
| 34 | //! one through every rename and edit that follows. |
| 35 | //! - A lightweight tag becomes an `Op::Mark` named for the tag and parented on |
| 36 | //! the tagged commit's own mark, which is exactly what a mark is for. |
| 37 | //! - Counters are a Lamport clock across all authors, so operations order as |
| 38 | //! the commits did whoever wrote them. |
| 39 | //! - A path is bytes throughout. Git permits a path no encoding will read, and |
| 40 | //! so does the vocabulary, so such a path imports like any other. |
| 41 | //! |
| 42 | //! # Every branch, and no branches |
| 43 | //! |
| 44 | //! Every branch in the repository is imported, because the log is a graph and can |
| 45 | //! hold them all. What is then written into the working copy is the render of the |
| 46 | //! whole log, which is the convergent merge of every branch and not the branch |
| 47 | //! `HEAD` points at. |
| 48 | //! |
| 49 | //! Ore has no branches to point at. The state is the operation set, and the |
| 50 | //! render of the whole set is what the history says the working copy is; write |
| 51 | //! one branch instead and the capture at the start of the very next verb records |
| 52 | //! the difference, asserting that branch as the truth and putting the others' |
| 53 | //! claims out of the way. That was a splice's worth of noise when a file was its |
| 54 | //! path, and it is a *file deletion* now that a file is an identity, because two |
| 55 | //! branches that each created `notes.md` are two files and only one of them was |
| 56 | //! written out. So the whole history is written out, `ore log` after an import |
| 57 | //! has nothing to capture, and a person who wants one branch's state asks for it |
| 58 | //! by name with `ore back`, every commit being a mark. |
| 59 | //! |
| 60 | //! # Two branches, one path |
| 61 | //! |
| 62 | //! A file is named by the operation that created it, so two branches that |
| 63 | //! independently created `notes.md` minted two files and both of them exist. That |
| 64 | //! is a state the repository can genuinely be in, and [`crate::tree::Layout`] |
| 65 | //! knows how to lay it out. It is not, however, a state a *git* history describes: |
| 66 | //! git merged the two into one blob at one path, and it has no second file to |
| 67 | //! speak of. So where a merge brings two files to one path, the import records an |
| 68 | //! `Op::FileDelete` for every one of them but the file the first parent held -- |
| 69 | //! the losing file keeps its bytes and its history, and stops claiming a path git |
| 70 | //! says it no longer has. Left alone it would clash for ever, in a repository |
| 71 | //! whose source never had a clash in it. |
| 72 | //! |
| 73 | //! # What is refused rather than guessed at |
| 74 | //! |
| 75 | //! The parser refuses a line it cannot place rather than reading past it, and |
| 76 | //! the import keeps the same posture. A submodule and a bare tree entry name |
| 77 | //! something that is not this repository's file, and an annotated tag carries a |
| 78 | //! message, a tagger and possibly a signature that no operation can hold. Each |
| 79 | //! is refused with the path or the name it concerns, because a history that |
| 80 | //! imported with a hole in it and nothing said would be a history the mirror |
| 81 | //! could never write back out. |
| 82 | //! |
| 83 | //! # What the import does not record |
| 84 | //! |
| 85 | //! - A `C` copy, which is recorded as a create of the destination holding the |
| 86 | //! source's bytes rather than as a copy. Git does not record a copy in a |
| 87 | //! commit either -- a commit is its tree, its parents and its metadata -- so |
| 88 | //! nothing a mirror needs is lost by it. |
| 89 | //! - A commit note, which annotates a commit and says nothing about the tree. |
| 90 | //! - The committer, where a commit names an author as well. Git's own tools |
| 91 | //! read a history by its authors, and only one of the two can be a replica. |
| 92 | //! |
| 93 | //! # What it costs to run |
| 94 | //! |
| 95 | //! Nothing whose size is the history is held for the length of the import. A |
| 96 | //! blob goes when the last change naming it has been read, a commit's state goes |
| 97 | //! when the last commit that can build on it has been translated, and a merge |
| 98 | //! parent other than the first is carried as operations and paths rather than as |
| 99 | //! a whole rendered repository. What remains resident is the log itself, which is |
| 100 | //! the repository, and one state -- one sequence and one render -- for the commit |
| 101 | //! in hand. Both grow with the history, and neither is the import's to let go. |
| 102 | |
| 103 | use crate::capture; |
| 104 | use crate::guard; |
| 105 | use crate::repo::{ |
| 106 | author_replica, |
| 107 | Repo, |
| 108 | CONFIG_FILE, |
| 109 | }; |
| 110 | use crate::tree; |
| 111 | |
| 112 | use oxedyne_fe2o3_core::prelude::*; |
| 113 | use oxedyne_fe2o3_text::secret; |
| 114 | use oxedyne_fe2o3_ore::fastexport::{ |
| 115 | BlobRef, |
| 116 | Commit, |
| 117 | Event, |
| 118 | FileChange, |
| 119 | FileMode, |
| 120 | ObjRef, |
| 121 | Parser, |
| 122 | Person, |
| 123 | }; |
| 124 | use oxedyne_fe2o3_ore::id::{ |
| 125 | OpId, |
| 126 | ReplicaId, |
| 127 | }; |
| 128 | use oxedyne_fe2o3_ore::log::Causality; |
| 129 | use oxedyne_fe2o3_ore::op::{ |
| 130 | with_author, |
| 131 | Mode, |
| 132 | Op, |
| 133 | }; |
| 134 | use oxedyne_fe2o3_ore::seq::render::Repo as Render; |
| 135 | use oxedyne_fe2o3_ore::seq::{ |
| 136 | OpOrder, |
| 137 | Sequence, |
| 138 | }; |
| 139 | |
| 140 | use std::collections::{ |
| 141 | BTreeMap, |
| 142 | BTreeSet, |
| 143 | HashMap, |
| 144 | }; |
| 145 | use std::io::{ |
| 146 | Read, |
| 147 | Write, |
| 148 | }; |
| 149 | use std::path::Path; |
| 150 | use std::process::{ |
| 151 | Command, |
| 152 | Stdio, |
| 153 | }; |
| 154 | |
| 155 | |
| 156 | /// How much of the stream is read at a time. |
| 157 | const CHUNK: usize = 64 * 1024; |
| 158 | |
| 159 | |
| 160 | /// The repository one commit leaves behind. |
| 161 | /// |
| 162 | /// There is one sequence, not one per file: which file a placement lands in is |
| 163 | /// read off the forest the render lays out, so a repository is rendered whole or |
| 164 | /// not at all. |
| 165 | #[derive(Clone, Default)] |
| 166 | struct State { |
| 167 | /// Every operation this commit's history holds. |
| 168 | seq: Sequence, |
| 169 | /// What they render as, as of the last render. |
| 170 | repo: Render, |
| 171 | /// Which file is live at each path, which is the tree git would check out. |
| 172 | at: BTreeMap<Vec<u8>, OpId>, |
| 173 | /// What each live file is, so that a mode already asserted is not asserted |
| 174 | /// again by every commit that leaves the file alone. |
| 175 | modes: BTreeMap<OpId, Mode>, |
| 176 | } |
| 177 | |
| 178 | /// What a merge takes from a parent other than the first. |
| 179 | /// |
| 180 | /// A merge wants two things of its other parents: their operations, so the union |
| 181 | /// can be rendered, and which file each of them had at each path, so the file the |
| 182 | /// first parent held is the one that keeps a contested path. It does not want |
| 183 | /// their renders. Every one of them is thrown away by the re-render that absorbing |
| 184 | /// forces, and a render is the larger half of a [`State`] -- the bytes of every |
| 185 | /// live file, and an atom for every insertion the history holds. Carrying one per |
| 186 | /// merge parent is what made the peak of an import fall on its merge commits. |
| 187 | struct Side { |
| 188 | seq: Sequence, // every operation its history holds |
| 189 | at: BTreeMap<Vec<u8>, OpId>, // which file it had at each path |
| 190 | } |
| 191 | |
| 192 | /// What a commit contributed to the log. |
| 193 | struct Done { |
| 194 | /// The mark operation that stands for the commit. |
| 195 | /// |
| 196 | /// Kept for the whole import, because a reference or a tag may name this |
| 197 | /// commit long after its state has been let go, and an `OpId` is a few bytes. |
| 198 | head: OpId, |
| 199 | /// The repository the commit leaves behind, while any later commit can still |
| 200 | /// build on it. |
| 201 | /// |
| 202 | /// `None` once the last commit that named this one as a parent has been |
| 203 | /// translated. See [`Import::needed`]: holding every commit's state for the |
| 204 | /// whole import is what made an import of a mid-sized repository take |
| 205 | /// gigabytes, since a `State` carries the entire sequence and the entire |
| 206 | /// rendered tree. |
| 207 | state: Option<State>, |
| 208 | } |
| 209 | |
| 210 | /// Everything the import is carrying while it reads the stream. |
| 211 | #[derive(Default)] |
| 212 | struct Import { |
| 213 | /// Blob payloads, by the mark the stream gave them. |
| 214 | blobs: HashMap<u64, Vec<u8>>, |
| 215 | wanted: HashMap<u64, usize>, // changes still to come naming each blob mark |
| 216 | /// Commits already translated, by the mark the stream gave them. |
| 217 | commits: HashMap<u64, Done>, |
| 218 | /// A second mark for an object already named. |
| 219 | aliases: HashMap<u64, u64>, |
| 220 | /// Where each reference stands, by the mark of the commit it points at. |
| 221 | refs: HashMap<String, u64>, |
| 222 | /// Every operation's parents, which is the graph a render is judged by. |
| 223 | parents: BTreeMap<OpId, Vec<OpId>>, |
| 224 | /// How many commits still to come name each mark as a parent. |
| 225 | /// |
| 226 | /// Filled by a first pass over the same stream and counted down by the second, |
| 227 | /// so that a commit's state can be let go the moment nothing can build on it. |
| 228 | /// Without this the import keeps one whole repository -- sequence, render and |
| 229 | /// all -- per commit it has read, which is why importing a repository of a few |
| 230 | /// hundred commits took gigabytes and was killed by the kernel rather than |
| 231 | /// finishing. |
| 232 | needed: HashMap<u64, usize>, |
| 233 | /// Whether this pass is the counting one, which resolves marks and parents |
| 234 | /// exactly as the real pass does and builds no state at all. |
| 235 | /// |
| 236 | /// The same code path deliberately: parent resolution depends on the marks, |
| 237 | /// the aliases and the reference table as they stand at that point in the |
| 238 | /// stream, so a census computed by separate logic could disagree with the pass |
| 239 | /// it is meant to describe, and would do so only on the histories complicated |
| 240 | /// enough to matter. |
| 241 | counting: bool, |
| 242 | /// How many commits were translated. |
| 243 | count: usize, |
| 244 | /// Marks minted for commits that the stream gave no mark, counted down from |
| 245 | /// the top so that they cannot collide with the stream's own. |
| 246 | minted: u64, |
| 247 | /// How many files a merge deleted for holding a path another file kept. |
| 248 | resolved: usize, |
| 249 | /// How many mode assertions the import recorded. |
| 250 | modes: usize, |
| 251 | /// The tags already recorded, so that a tag the stream names twice is one |
| 252 | /// mark and not two. |
| 253 | tagged: BTreeSet<(String, u64)>, |
| 254 | /// How many lightweight tags became marks. |
| 255 | tags: usize, |
| 256 | // Credentials in the history |
| 257 | // Gathered on the counting pass, where the bytes arrive and nothing has been |
| 258 | // authored. `found` holds only a blob that matched, so a history with nothing |
| 259 | // in it holds nothing here, and the bytes are dropped either way. |
| 260 | found: BTreeMap<u64, Vec<secret::Find>>, |
| 261 | carried: Vec<Carried>, |
| 262 | } |
| 263 | |
| 264 | /// A credential a commit of the imported history carries. |
| 265 | struct Carried { |
| 266 | mark: String, // the name the commit's mark takes |
| 267 | path: Vec<u8>, |
| 268 | line: usize, |
| 269 | kind: secret::Kind, |
| 270 | } |
| 271 | |
| 272 | |
| 273 | /// Runs git in a directory and returns its standard output, trimmed. |
| 274 | fn git(dir: &Path, args: &[&str]) |
| 275 | -> Outcome<String> |
| 276 | { |
| 277 | let out = res!(git_bytes(dir, args)); |
| 278 | Ok(fmt!("{}", String::from_utf8_lossy(&out).trim())) |
| 279 | } |
| 280 | |
| 281 | /// Runs git in a directory and returns its standard output as it came. |
| 282 | /// |
| 283 | /// Bytes, for the commands whose output holds a path: git does not require a |
| 284 | /// path to be text and neither does the vocabulary. |
| 285 | fn git_bytes(dir: &Path, args: &[&str]) |
| 286 | -> Outcome<Vec<u8>> |
| 287 | { |
| 288 | let out = match Command::new("git").current_dir(dir).args(args).output() { |
| 289 | Ok(o) => o, |
| 290 | Err(e) => return Err(err!(e, |
| 291 | "`git {}` could not be run in {:?}; is git on the path?", |
| 292 | args.join(" "), dir; |
| 293 | IO, System)), |
| 294 | }; |
| 295 | if !out.status.success() { |
| 296 | return Err(err!( |
| 297 | "`git {}` failed in {:?}: {}", args.join(" "), dir, |
| 298 | String::from_utf8_lossy(&out.stderr).trim(); |
| 299 | IO, System)); |
| 300 | } |
| 301 | Ok(out.stdout) |
| 302 | } |
| 303 | |
| 304 | /// Refuses the import where a branch tip still holds a credential. |
| 305 | /// |
| 306 | /// Ahead of both passes, and ahead of the first operation, because an operation |
| 307 | /// reaches the segment the moment it is authored and a store does not unwind: a |
| 308 | /// check after the walk would be a refusal with the whole history already |
| 309 | /// written and signed. [`crate::capture`] does not run here at all -- an import |
| 310 | /// authors from the stream and never from the working copy -- so the guard has |
| 311 | /// to be put on this path by name or it is not on it. |
| 312 | /// |
| 313 | /// What is scanned is every branch tip, which is what the render of the imported |
| 314 | /// log comes to: a file alive at no tip is a file some branch deleted, and a file |
| 315 | /// in the render is alive at the tip of the branch it lives on. Tags are not |
| 316 | /// scanned, because a tag names a commit in the past and the past is reported |
| 317 | /// rather than refused -- see [`Import::carried`]. |
| 318 | fn refuse_tip_credentials(from: &Path) |
| 319 | -> Outcome<()> |
| 320 | { |
| 321 | let refs = res!(git(from, &["for-each-ref", "--format=%(refname)", "refs/heads/"])); |
| 322 | // One entry per blob, holding every path that blob is at: identical bytes |
| 323 | // find identical lines, so the scan is per object and the report is per path. |
| 324 | let mut at: BTreeMap<String, Vec<Vec<u8>>> = BTreeMap::new(); |
| 325 | for refname in refs.lines() { |
| 326 | let listed = res!(git_bytes(from, &["ls-tree", "-r", "-z", refname])); |
| 327 | for entry in listed.split(|b| *b == 0) { |
| 328 | let (head, path) = match entry.iter().position(|b| *b == b'\t') { |
| 329 | Some(tab) => (&entry[..tab], entry[tab + 1..].to_vec()), |
| 330 | None => continue, |
| 331 | }; |
| 332 | // `<mode> <type> <oid>`, and a tree or a submodule is not bytes. |
| 333 | let parts: Vec<&[u8]> = head.split(|b| *b == b' ').collect(); |
| 334 | match parts.get(1) { |
| 335 | Some(kind) if *kind == b"blob" => (), |
| 336 | _ => continue, |
| 337 | }; |
| 338 | let oid = match parts.get(2) { |
| 339 | Some(oid) => fmt!("{}", String::from_utf8_lossy(oid)), |
| 340 | None => continue, |
| 341 | }; |
| 342 | // Lockfiles and vendored trees carry long hashes that read like keys. |
| 343 | if secret::skip_path(&path) { |
| 344 | continue; |
| 345 | } |
| 346 | let held = at.entry(oid).or_default(); |
| 347 | if !held.contains(&path) { |
| 348 | held.push(path); |
| 349 | } |
| 350 | } |
| 351 | } |
| 352 | if at.is_empty() { |
| 353 | return Ok(()); |
| 354 | } |
| 355 | let oids: Vec<&String> = at.keys().collect(); |
| 356 | let bytes = res!(cat_file_batch(from, &oids)); |
| 357 | let mut caught = Vec::new(); |
| 358 | for (oid, data) in oids.iter().zip(bytes.iter()) { |
| 359 | let paths = match at.get(*oid) { |
| 360 | Some(paths) => paths, |
| 361 | None => continue, |
| 362 | }; |
| 363 | for path in paths { |
| 364 | // Nothing is in the history yet -- an import refuses a repository that |
| 365 | // already holds operations -- so every finding is one this would be |
| 366 | // introducing, and there is no `had` to pass. |
| 367 | caught.extend(guard::inspect(path, data, None)); |
| 368 | } |
| 369 | } |
| 370 | if caught.is_empty() { |
| 371 | return Ok(()); |
| 372 | } |
| 373 | let mut said = String::new(); |
| 374 | for c in &caught { |
| 375 | said.push_str(&fmt!("\n {}:{} {}", |
| 376 | String::from_utf8_lossy(&c.path), c.line, c.kind.label())); |
| 377 | } |
| 378 | Err(err!( |
| 379 | "a credential is in what this import was about to record, so nothing was \ |
| 380 | recorded:\n{}\n\n\ |
| 381 | Those are the bytes a branch tip holds now, not something only the history \ |
| 382 | carries, so the value is live until somebody revokes it. An Ore history only \ |
| 383 | grows -- there is no rewrite, no prune and no --force, and a sync hands what \ |
| 384 | is written to every replica and to the git mirror -- and an import runs once, \ |
| 385 | because it refuses a repository that already holds operations. There would be \ |
| 386 | no second attempt to fix it in.\n\n\ |
| 387 | The value is not printed above. Open the file in the git repository being \ |
| 388 | imported, revoke what it holds, take it out of the tip and commit that, and \ |
| 389 | import again. Removing it from the tip does not remove it from the git \ |
| 390 | history and is not meant to: what that buys is an Ore log that does not carry \ |
| 391 | it live.\n\n\ |
| 392 | Where a match is genuinely a fixture, put `{}` in a comment on that line or \ |
| 393 | on the one above it, which is the marker the git pre-commit hook takes as \ |
| 394 | well.{}", |
| 395 | said, secret::MARKER, guard::unmarkable(&caught); |
| 396 | Security, Key, Permanent)) |
| 397 | } |
| 398 | |
| 399 | /// Reads a list of git objects in one pass, in the order they were asked for. |
| 400 | /// |
| 401 | /// One process for the whole tree rather than one for each object: a repository |
| 402 | /// worth importing has thousands of files, and a spawn each is the difference |
| 403 | /// between a second and a minute. |
| 404 | /// |
| 405 | /// **The list is written from a thread of its own, and this is not tidiness.** |
| 406 | /// Writing the whole list and then reading the answers deadlocks: git fills its |
| 407 | /// output pipe long before a list of thousands of object names has been sent, |
| 408 | /// and then blocks writing while this blocks writing, neither reading. It |
| 409 | /// survives every small repository, because a tree whose answers fit in a pipe |
| 410 | /// buffer never blocks -- fe2o3's 1,785 files hung for ever where a two-file |
| 411 | /// fixture passed. |
| 412 | fn cat_file_batch(dir: &Path, oids: &[&String]) |
| 413 | -> Outcome<Vec<Vec<u8>>> |
| 414 | { |
| 415 | let mut child = match Command::new("git") |
| 416 | .current_dir(dir) |
| 417 | .args(["cat-file", "--batch"]) |
| 418 | .stdin(Stdio::piped()) |
| 419 | .stdout(Stdio::piped()) |
| 420 | .stderr(Stdio::piped()) |
| 421 | .spawn() |
| 422 | { |
| 423 | Ok(c) => c, |
| 424 | Err(e) => return Err(err!(e, |
| 425 | "`git cat-file --batch` could not be started in {:?}.", dir; |
| 426 | IO, System)), |
| 427 | }; |
| 428 | let mut asked = String::new(); |
| 429 | for oid in oids { |
| 430 | asked.push_str(oid); |
| 431 | asked.push('\n'); |
| 432 | } |
| 433 | let mut sink = match child.stdin.take() { |
| 434 | Some(sink) => sink, |
| 435 | None => return Err(err!( |
| 436 | "`git cat-file --batch` gave no standard input to write."; |
| 437 | IO, System, Missing)), |
| 438 | }; |
| 439 | let writing = std::thread::spawn(move || { |
| 440 | let wrote = sink.write_all(asked.as_bytes()); |
| 441 | // Dropped here, so that git sees the end of the list and finishes rather |
| 442 | // than waiting for more. |
| 443 | drop(sink); |
| 444 | wrote |
| 445 | }); |
| 446 | let out = match child.wait_with_output() { |
| 447 | Ok(o) => o, |
| 448 | Err(e) => return Err(err!(e, |
| 449 | "`git cat-file --batch` could not be read in {:?}.", dir; |
| 450 | IO, Read)), |
| 451 | }; |
| 452 | match writing.join() { |
| 453 | Ok(Ok(())) => (), |
| 454 | Ok(Err(e)) => return Err(err!(e, |
| 455 | "The object list could not be handed to `git cat-file --batch`."; |
| 456 | IO, Write)), |
| 457 | Err(_) => return Err(err!( |
| 458 | "The thread writing to `git cat-file --batch` did not finish."; |
| 459 | Bug, Thread)), |
| 460 | } |
| 461 | if !out.status.success() { |
| 462 | return Err(err!( |
| 463 | "`git cat-file --batch` failed in {:?}: {}", dir, |
| 464 | String::from_utf8_lossy(&out.stderr).trim(); |
| 465 | IO, System)); |
| 466 | } |
| 467 | // `<oid> <type> <size>` on a line, then that many bytes, then a line feed. |
| 468 | let mut got = Vec::new(); |
| 469 | let mut at = 0usize; |
| 470 | let data = out.stdout; |
| 471 | while at < data.len() { |
| 472 | let end = match data[at..].iter().position(|b| *b == b'\n') { |
| 473 | Some(n) => at + n, |
| 474 | None => break, |
| 475 | }; |
| 476 | let head = String::from_utf8_lossy(&data[at..end]).into_owned(); |
| 477 | let size: usize = match head.rsplit(' ').next().and_then(|n| n.parse().ok()) { |
| 478 | Some(n) => n, |
| 479 | None => return Err(err!( |
| 480 | "`git cat-file --batch` answered {:?}, which names no size.", head; |
| 481 | Invalid, Input, Mismatch)), |
| 482 | }; |
| 483 | let from = end + 1; |
| 484 | let to = from + size; |
| 485 | if to > data.len() { |
| 486 | return Err(err!( |
| 487 | "`git cat-file --batch` promised {} bytes for {:?} and gave {}.", |
| 488 | size, head, data.len() - from; |
| 489 | Invalid, Input, Mismatch)); |
| 490 | } |
| 491 | got.push(data[from..to].to_vec()); |
| 492 | at = to + 1; |
| 493 | } |
| 494 | if got.len() != oids.len() { |
| 495 | return Err(err!( |
| 496 | "{} objects were asked for and {} came back.", oids.len(), got.len(); |
| 497 | Bug, Mismatch)); |
| 498 | } |
| 499 | Ok(got) |
| 500 | } |
| 501 | |
| 502 | // How much of a long answer is shown |
| 503 | // A history nobody guarded can carry a great many findings, and a line each is a |
| 504 | // wall of text somebody scrolls past, which is the same as saying nothing. The |
| 505 | // totals are always exact; it is the listing that stops. |
| 506 | const FILES_SHOWN: usize = 20; |
| 507 | const LINES_SHOWN: usize = 5; |
| 508 | |
| 509 | /// Writes what the history carries to stderr, and returns the line stdout gets. |
| 510 | /// |
| 511 | /// Grouped by file and then by the line within it, because one credential |
| 512 | /// committed once and touched by forty later commits is one thing to fix and |
| 513 | /// forty findings, and the ungrouped form buries the two files that matter under |
| 514 | /// the one that does not. Each group says how many marks carry it and names the |
| 515 | /// first and the last, which is the span a person would go looking through. |
| 516 | /// |
| 517 | /// **The value is never here.** Not the bytes, not a prefix of them, not a |
| 518 | /// masked version: this text is read out of a terminal, pasted into a note and |
| 519 | /// may well end up on a forge page, and a report that carried the key into all |
| 520 | /// three would be the second way it escaped. |
| 521 | fn report_carried(carried: &[Carried]) -> Option<String> { |
| 522 | if carried.is_empty() { |
| 523 | return None; |
| 524 | } |
| 525 | // Path, then the line and kind within it, then how many marks and which the |
| 526 | // first and last were. Findings arrive in commit order, so first and last are |
| 527 | // the commit that introduced it and the last one still carrying it. |
| 528 | let mut by_file: BTreeMap<&[u8], BTreeMap<(usize, &str), (usize, &str, &str)>> = |
| 529 | BTreeMap::new(); |
| 530 | let mut marks: BTreeSet<&str> = BTreeSet::new(); |
| 531 | for c in carried { |
| 532 | marks.insert(&c.mark); |
| 533 | let at = by_file.entry(&c.path).or_default(); |
| 534 | match at.get_mut(&(c.line, c.kind.label())) { |
| 535 | Some(held) => { |
| 536 | held.0 += 1; |
| 537 | held.2 = &c.mark; |
| 538 | }, |
| 539 | None => { |
| 540 | at.insert((c.line, c.kind.label()), (1, &c.mark, &c.mark)); |
| 541 | }, |
| 542 | } |
| 543 | } |
| 544 | eprintln!("ore: a credential is in the history this import recorded, and an Ore \ |
| 545 | history only grows:"); |
| 546 | for (n, (path, lines)) in by_file.iter().enumerate() { |
| 547 | if n == FILES_SHOWN { |
| 548 | eprintln!(" ... and {} more file{}", |
| 549 | by_file.len() - FILES_SHOWN, |
| 550 | if by_file.len() - FILES_SHOWN == 1 { "" } else { "s" }); |
| 551 | break; |
| 552 | } |
| 553 | eprintln!(" {}", tree::shown(path)); |
| 554 | for (m, ((line, kind), (count, first, last))) in lines.iter().enumerate() { |
| 555 | if m == LINES_SHOWN { |
| 556 | eprintln!(" ... and {} more finding{} in this file", |
| 557 | lines.len() - LINES_SHOWN, |
| 558 | if lines.len() - LINES_SHOWN == 1 { "" } else { "s" }); |
| 559 | break; |
| 560 | } |
| 561 | if first == last { |
| 562 | eprintln!(" {} at line {}, in the mark {:?}", kind, line, first); |
| 563 | } else { |
| 564 | eprintln!(" {} at line {}, in {} marks, from {:?} to {:?}", |
| 565 | kind, line, count, first, last); |
| 566 | } |
| 567 | } |
| 568 | } |
| 569 | eprintln!(" {} finding{} over {} file{} and {} mark{}. The values are not \ |
| 570 | printed: open the files in the git repository to see them.", |
| 571 | carried.len(), if carried.len() == 1 { "" } else { "s" }, |
| 572 | by_file.len(), if by_file.len() == 1 { "" } else { "s" }, |
| 573 | marks.len(), if marks.len() == 1 { "" } else { "s" }); |
| 574 | eprintln!(" These are what the history holds, not what the tip holds -- a tip \ |
| 575 | still carrying one is refused instead. Revoke anything still live: removing \ |
| 576 | a file removes it from neither history, and this one is now signed."); |
| 577 | Some(fmt!("{} credential finding{} in the history, over {} file{} and {} mark{}, \ |
| 578 | listed on stderr", |
| 579 | carried.len(), if carried.len() == 1 { "" } else { "s" }, |
| 580 | by_file.len(), if by_file.len() == 1 { "" } else { "s" }, |
| 581 | marks.len(), if marks.len() == 1 { "" } else { "s" })) |
| 582 | } |
| 583 | |
| 584 | /// Returns the identity line an author is known by. |
| 585 | /// |
| 586 | /// This and not [`author_line`] is what keys the author map, because a person is |
| 587 | /// one author across every commit and a moment belongs to one commit. |
| 588 | fn identity(who: &Person) -> String { |
| 589 | fmt!("{} <{}>", who.name_lossy(), who.email_lossy()) |
| 590 | } |
| 591 | |
| 592 | /// Returns the author line as git itself writes one: identity, seconds, offset. |
| 593 | /// |
| 594 | /// Git's own spelling, because a mark can carry the whole of it in one line and |
| 595 | /// a reader taking one line off the end is the rule that keeps a body somebody |
| 596 | /// wrote from being mistaken for this. The offset is the part that is otherwise |
| 597 | /// lost: [`Op::Mark`] holds seconds in UTC and no zone, so `+0800` is in the |
| 598 | /// history only if it is here. |
| 599 | fn author_line(who: &Person) -> String { |
| 600 | fmt!("{} {}", identity(who), who.when) |
| 601 | } |
| 602 | |
| 603 | /// Returns the subject of a commit message, which is what the mark is named. |
| 604 | /// |
| 605 | /// Git's subject is not the first line of the message: it runs to the first |
| 606 | /// blank line, and a subject somebody wrapped over two lines folds onto one |
| 607 | /// joined by spaces. That is what `git log --format=%s` prints, so it is what |
| 608 | /// every git reader takes the commit to be called, and a mark named anything |
| 609 | /// else is a mark nobody can find the commit by. Taking the first physical line |
| 610 | /// instead cuts a wrapped subject off at the wrap, and every commit beginning |
| 611 | /// `fe2o3_social:` on a line of its own then comes out with the same name. |
| 612 | fn subject(message: &[u8]) -> String { |
| 613 | let text = String::from_utf8_lossy(message); |
| 614 | let mut lines: Vec<&str> = Vec::new(); |
| 615 | for line in text.lines() { |
| 616 | // Trailing whitespace goes and leading whitespace stays, which is what |
| 617 | // git does, so an indented continuation keeps its indent as a second |
| 618 | // space. |
| 619 | let line = line.trim_end(); |
| 620 | if line.is_empty() { |
| 621 | break; |
| 622 | } |
| 623 | lines.push(line); |
| 624 | } |
| 625 | if lines.is_empty() { |
| 626 | fmt!("(no message)") |
| 627 | } else { |
| 628 | lines.join(" ") |
| 629 | } |
| 630 | } |
| 631 | |
| 632 | /// Returns what a commit message says after its subject, where it says anything. |
| 633 | /// |
| 634 | /// Bytes and not text, because a commit message is not required to be UTF-8 and |
| 635 | /// [`Op::Mark`] holds a body for that reason. They are the message's own bytes: |
| 636 | /// git's `%b` strips the trailing whitespace off every line it prints, and a |
| 637 | /// mirror written back from that would not be the history that was imported. |
| 638 | fn body(message: &[u8]) -> Option<Vec<u8>> { |
| 639 | let blank = |line: &[u8]| line.iter().all(|b| b.is_ascii_whitespace()); |
| 640 | let ends = |at: usize| message[at..].iter() |
| 641 | .position(|b| *b == b'\n') |
| 642 | .map(|n| at + n) |
| 643 | .unwrap_or(message.len()); |
| 644 | let mut at = 0; |
| 645 | while at < message.len() && !blank(&message[at..ends(at)]) { |
| 646 | at = (ends(at) + 1).min(message.len()); |
| 647 | } |
| 648 | while at < message.len() && blank(&message[at..ends(at)]) { |
| 649 | at = (ends(at) + 1).min(message.len()); |
| 650 | } |
| 651 | if at >= message.len() { |
| 652 | None |
| 653 | } else { |
| 654 | Some(message[at..].to_vec()) |
| 655 | } |
| 656 | } |
| 657 | |
| 658 | /// Returns the live file at each path of a render, one file to a path. |
| 659 | /// |
| 660 | /// A path two live files claim is decided the way [`crate::tree::Layout`] decides |
| 661 | /// it, by op order, so that the tree the import carries and the tree it writes out |
| 662 | /// never disagree about which file a name stands for. |
| 663 | fn live_at(render: &Render) -> BTreeMap<Vec<u8>, OpId> { |
| 664 | let mut out: BTreeMap<Vec<u8>, OpId> = BTreeMap::new(); |
| 665 | for file in render.live() { |
| 666 | let id = file.file(); |
| 667 | match out.get(file.path()) { |
| 668 | Some(seen) if OpOrder::of(seen) >= OpOrder::of(&id) => (), |
| 669 | _ => { |
| 670 | out.insert(file.path().to_vec(), id); |
| 671 | }, |
| 672 | } |
| 673 | } |
| 674 | out |
| 675 | } |
| 676 | |
| 677 | /// Returns what each live file of a render is, by identity. |
| 678 | fn modes_of(render: &Render) -> BTreeMap<OpId, Mode> { |
| 679 | render.live().into_iter().map(|f| (f.file(), f.mode())).collect() |
| 680 | } |
| 681 | |
| 682 | /// Returns the tag a reference names, where it names one. |
| 683 | /// |
| 684 | /// Git spells a lightweight tag as a reference under `refs/tags/` and nothing |
| 685 | /// else, so this is the whole of recognising one. |
| 686 | fn tag_name(refname: &str) -> Option<&str> { |
| 687 | refname.strip_prefix("refs/tags/") |
| 688 | } |
| 689 | |
| 690 | impl Import { |
| 691 | |
| 692 | /// Resolves a mark through any aliases the stream declared. |
| 693 | fn resolve(&self, mark: u64) -> u64 { |
| 694 | let mut at = mark; |
| 695 | for _ in 0..64 { |
| 696 | match self.aliases.get(&at) { |
| 697 | Some(next) if *next != at => at = *next, |
| 698 | _ => return at, |
| 699 | } |
| 700 | } |
| 701 | at |
| 702 | } |
| 703 | |
| 704 | /// Returns the bytes a file change points at, taking them where nothing later |
| 705 | /// wants them. |
| 706 | fn bytes_of(&mut self, data: &BlobRef) |
| 707 | -> Outcome<Vec<u8>> |
| 708 | { |
| 709 | match data { |
| 710 | BlobRef::Inline(bytes) => Ok(bytes.clone()), |
| 711 | BlobRef::Mark(m) => { |
| 712 | let at = self.resolve(*m); |
| 713 | // Moved out rather than copied where this is the last change that |
| 714 | // names it, which is both the copy saved and the blob let go. A mark |
| 715 | // the census never saw is one no change can name twice, so taking it |
| 716 | // is safe there too. |
| 717 | let last = match self.wanted.get_mut(&at) { |
| 718 | Some(count) => { |
| 719 | *count = count.saturating_sub(1); |
| 720 | *count == 0 |
| 721 | }, |
| 722 | None => true, |
| 723 | }; |
| 724 | let held = if last { |
| 725 | self.blobs.remove(&at) |
| 726 | } else { |
| 727 | self.blobs.get(&at).cloned() |
| 728 | }; |
| 729 | match held { |
| 730 | Some(b) => Ok(b), |
| 731 | None => Err(err!( |
| 732 | "The stream names the blob :{}, which it has not described, or \ |
| 733 | whose bytes were let go before this change asked for them. The \ |
| 734 | census counted the changes naming it wrongly.", m; |
| 735 | Invalid, Input, Missing)), |
| 736 | } |
| 737 | }, |
| 738 | BlobRef::Name(name) => Err(err!( |
| 739 | "The stream names the object {:?} rather than a mark, so its bytes \ |
| 740 | are not in the stream.", name; |
| 741 | Invalid, Input, Missing)), |
| 742 | } |
| 743 | } |
| 744 | |
| 745 | /// Renders the repository against the whole operation graph the import has |
| 746 | /// built. |
| 747 | fn render(&self, seq: &Sequence) |
| 748 | -> Outcome<Render> |
| 749 | { |
| 750 | let cause = Causality::new(self.parents.iter().map(|(id, p)| (*id, p.as_slice()))); |
| 751 | let render = match seq.render_with(&cause) { |
| 752 | Ok(r) => r, |
| 753 | Err(e) => return Err(err!(e, |
| 754 | "The imported repository of {} operations could not be rendered.", |
| 755 | seq.len(); |
| 756 | Invalid, Data)), |
| 757 | }; |
| 758 | Ok(render) |
| 759 | } |
| 760 | |
| 761 | /// Appends one operation, remembers its parents, puts it in the state the |
| 762 | /// commit is building, and makes it the head of the chain. |
| 763 | fn mint( |
| 764 | &mut self, |
| 765 | repo: &mut Repo, |
| 766 | state: &mut State, |
| 767 | replica: ReplicaId, |
| 768 | chain: &mut Vec<OpId>, |
| 769 | op: Op, |
| 770 | ) |
| 771 | -> Outcome<OpId> |
| 772 | { |
| 773 | let id = res!(repo.author_with(replica, chain.clone(), op)); |
| 774 | self.parents.insert(id, chain.clone()); |
| 775 | let rec = match repo.log.get(&id) { |
| 776 | Some(r) => r.clone(), |
| 777 | None => return Err(err!( |
| 778 | "The operation {} was appended and is not in the log.", id; |
| 779 | Bug, Missing)), |
| 780 | }; |
| 781 | res!(state.seq.apply_record(&rec)); |
| 782 | *chain = vec![id]; |
| 783 | Ok(id) |
| 784 | } |
| 785 | |
| 786 | /// Brings the other parents of a merge into the state the first parent left. |
| 787 | /// |
| 788 | /// A merge is where two histories meet, and what the merge leaves behind has |
| 789 | /// to be built from both of them: the operations of every parent, in one |
| 790 | /// sequence, rendered together. What that renders as is the convergent merge, |
| 791 | /// flags and all -- and it is very often not what the person who resolved the |
| 792 | /// merge in git chose, which is why the caller then splices it to the bytes git |
| 793 | /// recorded. Both facts end up in the history: the concurrency, and the |
| 794 | /// resolution. |
| 795 | /// |
| 796 | /// Where the union leaves two live files at one path, the file the first parent |
| 797 | /// held keeps the path and the others are deleted. See the module note. |
| 798 | fn union( |
| 799 | &mut self, |
| 800 | repo: &mut Repo, |
| 801 | state: &mut State, |
| 802 | others: Vec<Side>, |
| 803 | replica: ReplicaId, |
| 804 | chain: &mut Vec<OpId>, |
| 805 | ) |
| 806 | -> Outcome<()> |
| 807 | { |
| 808 | if others.is_empty() { |
| 809 | return Ok(()); |
| 810 | } |
| 811 | // Which file each path is preferred to hold: the first parent decides, and |
| 812 | // then the others in the order git listed them. |
| 813 | let mut prefer = state.at.clone(); |
| 814 | let mut fresh = false; |
| 815 | for other in &others { |
| 816 | if res!(state.seq.absorb(&other.seq)) > 0 { |
| 817 | fresh = true; |
| 818 | } |
| 819 | for (path, id) in &other.at { |
| 820 | prefer.entry(path.clone()).or_insert(*id); |
| 821 | } |
| 822 | } |
| 823 | // Absorbed and read, so the parents go before the render they made necessary |
| 824 | // is built. |
| 825 | drop(others); |
| 826 | if !fresh { |
| 827 | return Ok(()); |
| 828 | } |
| 829 | // The render standing here is let go before its replacement is built. The |
| 830 | // replacement is built from the sequence and never reads it, so assigning |
| 831 | // over it instead holds two renders of the whole repository at once, which |
| 832 | // is where the peak of an import used to sit. |
| 833 | state.repo = Render::default(); |
| 834 | state.repo = res!(self.render(&state.seq)); |
| 835 | let clashes: Vec<(Vec<u8>, Vec<OpId>)> = state.repo.clashes() |
| 836 | .into_iter() |
| 837 | .map(|(path, ids)| (path.to_vec(), ids)) |
| 838 | .collect(); |
| 839 | if clashes.is_empty() { |
| 840 | state.at = live_at(&state.repo); |
| 841 | state.modes = modes_of(&state.repo); |
| 842 | return Ok(()); |
| 843 | } |
| 844 | for (path, ids) in clashes { |
| 845 | let kept = match prefer.get(&path) { |
| 846 | Some(id) if ids.contains(id) => *id, |
| 847 | _ => match ids.iter().copied().max_by_key(OpOrder::of) { |
| 848 | Some(id) => id, |
| 849 | None => continue, |
| 850 | }, |
| 851 | }; |
| 852 | for id in ids { |
| 853 | if id == kept { |
| 854 | continue; |
| 855 | } |
| 856 | res!(self.mint(repo, state, replica, chain, Op::FileDelete { file: id })); |
| 857 | self.resolved += 1; |
| 858 | } |
| 859 | } |
| 860 | state.repo = Render::default(); |
| 861 | state.repo = res!(self.render(&state.seq)); |
| 862 | state.at = live_at(&state.repo); |
| 863 | state.modes = modes_of(&state.repo); |
| 864 | Ok(()) |
| 865 | } |
| 866 | |
| 867 | /// Puts bytes at a path, minting whatever operations that calls for. |
| 868 | /// |
| 869 | /// A file that is not there is created, and the splice that fills it anchors at |
| 870 | /// the origin anchor of the create, which is what says which file it lands in. |
| 871 | /// A file that is there is spliced against the render the commit began with, |
| 872 | /// which is sound because a splice into one file cannot change what another |
| 873 | /// file's render says. |
| 874 | /// |
| 875 | /// The mode is asserted only where it differs from what the file already is. |
| 876 | /// Git states a mode against every path of every tree, so recording one per |
| 877 | /// commit per file would bury the history in operations saying nothing; a |
| 878 | /// file is normal until something says otherwise, and after that it is |
| 879 | /// whatever the last assertion said. |
| 880 | #[allow(clippy::too_many_arguments)] |
| 881 | fn put( |
| 882 | &mut self, |
| 883 | repo: &mut Repo, |
| 884 | state: &mut State, |
| 885 | chain: &mut Vec<OpId>, |
| 886 | replica: ReplicaId, |
| 887 | path: &[u8], |
| 888 | mode: Mode, |
| 889 | bytes: &[u8], |
| 890 | ) |
| 891 | -> Outcome<()> |
| 892 | { |
| 893 | let file = match state.at.get(path).copied() { |
| 894 | Some(id) => id, |
| 895 | None => { |
| 896 | let id = res!(self.mint(repo, state, replica, chain, |
| 897 | Op::FileCreate { path: path.to_vec() })); |
| 898 | state.at.insert(path.to_vec(), id); |
| 899 | state.modes.insert(id, Mode::default()); |
| 900 | if !bytes.is_empty() { |
| 901 | res!(self.mint(repo, state, replica, chain, |
| 902 | capture::fill(id, bytes.to_vec()))); |
| 903 | } |
| 904 | res!(self.set_mode(repo, state, replica, chain, id, mode)); |
| 905 | return Ok(()); |
| 906 | }, |
| 907 | }; |
| 908 | let view = match state.repo.file(file) { |
| 909 | Some(v) => v, |
| 910 | None => return Err(err!( |
| 911 | "The file {} sits at {:?} and the render does not hold it.", |
| 912 | file, tree::shown(path); |
| 913 | Bug, Missing)), |
| 914 | }; |
| 915 | // Every splice is resolved against the render the commit began with before |
| 916 | // any of them is applied, which is what the difference reports its offsets |
| 917 | // against. |
| 918 | let ops = res!(capture::derive(view, bytes)); |
| 919 | for op in ops { |
| 920 | res!(self.mint(repo, state, replica, chain, op)); |
| 921 | } |
| 922 | res!(self.set_mode(repo, state, replica, chain, file, mode)); |
| 923 | Ok(()) |
| 924 | } |
| 925 | |
| 926 | /// Records what a file is, where that is not what it already was. |
| 927 | fn set_mode( |
| 928 | &mut self, |
| 929 | repo: &mut Repo, |
| 930 | state: &mut State, |
| 931 | replica: ReplicaId, |
| 932 | chain: &mut Vec<OpId>, |
| 933 | file: OpId, |
| 934 | mode: Mode, |
| 935 | ) |
| 936 | -> Outcome<()> |
| 937 | { |
| 938 | let was = state.modes.get(&file).copied().unwrap_or_default(); |
| 939 | if was == mode { |
| 940 | return Ok(()); |
| 941 | } |
| 942 | res!(self.mint(repo, state, replica, chain, Op::FileMode { file, mode })); |
| 943 | state.modes.insert(file, mode); |
| 944 | self.modes += 1; |
| 945 | Ok(()) |
| 946 | } |
| 947 | |
| 948 | /// Records a lightweight tag as a mark on the commit it names. |
| 949 | /// |
| 950 | /// The mark is parented on the tagged commit's own mark and on nothing else, |
| 951 | /// so a tag says where in the history it points and adds no edge the git |
| 952 | /// history did not have. It is authored by the replica that wrote the commit, |
| 953 | /// there being nobody else the stream names: a lightweight tag is a reference |
| 954 | /// and carries no tagger. |
| 955 | /// |
| 956 | /// The stream may name one tag twice -- once as the reference a commit was |
| 957 | /// written to, once as a `reset` -- so a tag already recorded is not recorded |
| 958 | /// again. |
| 959 | fn lightweight_tag(&mut self, repo: &mut Repo, name: &str, mark: u64) |
| 960 | -> Outcome<()> |
| 961 | { |
| 962 | if !self.tagged.insert((fmt!("{}", name), mark)) { |
| 963 | return Ok(()); |
| 964 | } |
| 965 | let head = match self.commits.get(&mark) { |
| 966 | Some(done) => done.head, |
| 967 | // A tag on something that is not a commit in this stream -- a tag of a |
| 968 | // blob, or of a commit the export left out -- points at nothing this |
| 969 | // history can name. |
| 970 | None => return Ok(()), |
| 971 | }; |
| 972 | let id = res!(repo.author_with( |
| 973 | head.replica, |
| 974 | vec![head], |
| 975 | Op::Mark { name: fmt!("{}", name), body: None, time: None }, |
| 976 | )); |
| 977 | self.parents.insert(id, vec![head]); |
| 978 | self.tags += 1; |
| 979 | Ok(()) |
| 980 | } |
| 981 | |
| 982 | /// Translates one commit into operations. |
| 983 | fn commit(&mut self, repo: &mut Repo, commit: &Commit) |
| 984 | -> Outcome<()> |
| 985 | { |
| 986 | // The author where a commit names one, and the committer where it does |
| 987 | // not, which is what git itself falls back to. Both the replica and the |
| 988 | // date come off this one line, so the two can never disagree about which |
| 989 | // of the two people a commit names it was written by. |
| 990 | let by = commit.author.as_ref().unwrap_or(&commit.committer); |
| 991 | let who = identity(by); |
| 992 | // A mapping already in the configuration wins, and the derivation is the |
| 993 | // fallback rather than the rule. One person is often several identity |
| 994 | // lines -- a name that changed, an address that lapsed, a machine |
| 995 | // configured once and forgotten -- and deriving a replica from each of |
| 996 | // them imports one author as several. The mapping is the only chance to |
| 997 | // say so, because a replica identifier is part of every operation |
| 998 | // identifier it writes, is covered by the signature over the header, and |
| 999 | // is therefore not something a later pass can correct. See `authors` in |
| 1000 | // [`crate::repo::Config`]. |
| 1001 | let replica = match repo.cfg.authors.get(&who) { |
| 1002 | Some(known) => ReplicaId::new(*known), |
| 1003 | None => { |
| 1004 | let derived = author_replica(&who); |
| 1005 | repo.cfg.authors.insert(who.clone(), derived.inner()); |
| 1006 | derived |
| 1007 | }, |
| 1008 | }; |
| 1009 | // The stream omits `from` when the branch already stands where the commit |
| 1010 | // builds on, so the reference is the fallback and not an error. |
| 1011 | let first = match &commit.from { |
| 1012 | Some(ObjRef::Mark(m)) => Some(self.resolve(*m)), |
| 1013 | Some(ObjRef::Name(n)) => return Err(err!( |
| 1014 | "The commit on {:?} names the parent {:?} rather than a mark.", |
| 1015 | commit.refname, n; |
| 1016 | Invalid, Input, Missing)), |
| 1017 | None => self.refs.get(&commit.refname).copied(), |
| 1018 | }; |
| 1019 | let mut parent_marks: Vec<u64> = Vec::new(); |
| 1020 | if let Some(m) = first { |
| 1021 | parent_marks.push(m); |
| 1022 | } |
| 1023 | for merge in &commit.merges { |
| 1024 | match merge { |
| 1025 | ObjRef::Mark(m) => parent_marks.push(self.resolve(*m)), |
| 1026 | ObjRef::Name(n) => return Err(err!( |
| 1027 | "The commit on {:?} names the merge parent {:?} rather than a \ |
| 1028 | mark.", commit.refname, n; |
| 1029 | Invalid, Input, Missing)), |
| 1030 | } |
| 1031 | } |
| 1032 | // The counting pass ends here. Everything above resolved which marks this |
| 1033 | // commit builds on, which is the whole of what the census wants; everything |
| 1034 | // below builds the state that the census exists to avoid building. |
| 1035 | if self.counting { |
| 1036 | for m in &parent_marks { |
| 1037 | *self.needed.entry(*m).or_insert(0) += 1; |
| 1038 | } |
| 1039 | // The blobs are counted here for the same reason the parents are: a |
| 1040 | // blob held to the end of the import is every version of every file |
| 1041 | // the history ever had, resident at once, and nothing reads it twice. |
| 1042 | // Counted per change and not per commit, because one commit writing |
| 1043 | // one blob to two paths wants its bytes twice. |
| 1044 | for change in &commit.changes { |
| 1045 | if let FileChange::Modify { data: BlobRef::Mark(m), .. } = change { |
| 1046 | let at = self.resolve(*m); |
| 1047 | *self.wanted.entry(at).or_insert(0) += 1; |
| 1048 | } |
| 1049 | } |
| 1050 | // Which commit put a credential in, by the name its mark will take. A |
| 1051 | // blob named by object name rather than by mark is not looked at, |
| 1052 | // because `bytes_of` refuses one outright and no import carrying one |
| 1053 | // ever finishes. |
| 1054 | let named = subject(&commit.message); |
| 1055 | for change in &commit.changes { |
| 1056 | let (data, path) = match change { |
| 1057 | FileChange::Modify { data, path, .. } => (data, path), |
| 1058 | _ => continue, |
| 1059 | }; |
| 1060 | if secret::skip_path(path) { |
| 1061 | continue; |
| 1062 | } |
| 1063 | let found = match data { |
| 1064 | BlobRef::Mark(m) => { |
| 1065 | let at = self.resolve(*m); |
| 1066 | self.found.get(&at).cloned().unwrap_or_default() |
| 1067 | }, |
| 1068 | BlobRef::Inline(bytes) => secret::scan(bytes), |
| 1069 | BlobRef::Name(_) => Vec::new(), |
| 1070 | }; |
| 1071 | for find in found { |
| 1072 | self.carried.push(Carried { |
| 1073 | mark: named.clone(), |
| 1074 | path: path.clone(), |
| 1075 | line: find.line, |
| 1076 | kind: find.kind, |
| 1077 | }); |
| 1078 | } |
| 1079 | } |
| 1080 | let mark = match commit.mark { |
| 1081 | Some(m) => m, |
| 1082 | None => { |
| 1083 | self.minted += 1; |
| 1084 | u64::MAX - self.minted |
| 1085 | }, |
| 1086 | }; |
| 1087 | self.refs.insert(commit.refname.clone(), mark); |
| 1088 | self.count += 1; |
| 1089 | return Ok(()); |
| 1090 | } |
| 1091 | let mut chain: Vec<OpId> = Vec::new(); |
| 1092 | let mut state = State::default(); |
| 1093 | let mut others: Vec<Side> = Vec::new(); |
| 1094 | for (n, m) in parent_marks.iter().enumerate() { |
| 1095 | // Taken rather than borrowed where this is the last commit that can |
| 1096 | // name it, so the state moves into this commit instead of being copied |
| 1097 | // beside it. The clone that remains is the one a parent with other |
| 1098 | // children still owes. |
| 1099 | let last = match self.needed.get_mut(m) { |
| 1100 | Some(count) => { |
| 1101 | *count = count.saturating_sub(1); |
| 1102 | *count == 0 |
| 1103 | }, |
| 1104 | // A parent nothing counted is a parent nothing else can name. |
| 1105 | None => true, |
| 1106 | }; |
| 1107 | let done = match self.commits.get_mut(m) { |
| 1108 | Some(d) => d, |
| 1109 | None => return Err(err!( |
| 1110 | "The commit on {:?} names the parent :{}, which the stream has \ |
| 1111 | not described.", commit.refname, m; |
| 1112 | Invalid, Input, Missing)), |
| 1113 | }; |
| 1114 | chain.push(done.head); |
| 1115 | // The first parent is the one this commit is a difference from, so it is |
| 1116 | // wanted whole. A later parent is wanted only for what a merge reads of |
| 1117 | // it, and taking that much is what keeps a merge from carrying a render |
| 1118 | // per parent. Either is taken rather than borrowed where this is the last |
| 1119 | // commit that can name it, so the state moves into this commit instead of |
| 1120 | // being copied beside it. The clone that remains is the one a parent with |
| 1121 | // other children still owes. |
| 1122 | let gone = || err!( |
| 1123 | "The commit on {:?} names the parent :{}, whose state was let go \ |
| 1124 | before this commit asked for it. The census counted the parents \ |
| 1125 | wrongly.", commit.refname, m; |
| 1126 | Bug, Missing); |
| 1127 | if n == 0 { |
| 1128 | state = match if last { done.state.take() } else { done.state.clone() } { |
| 1129 | Some(st) => st, |
| 1130 | None => return Err(gone()), |
| 1131 | }; |
| 1132 | } else if last { |
| 1133 | match done.state.take() { |
| 1134 | Some(st) => others.push(Side { seq: st.seq, at: st.at }), |
| 1135 | None => return Err(gone()), |
| 1136 | } |
| 1137 | } else { |
| 1138 | match &done.state { |
| 1139 | Some(st) => others.push(Side { |
| 1140 | seq: st.seq.clone(), |
| 1141 | at: st.at.clone(), |
| 1142 | }), |
| 1143 | None => return Err(gone()), |
| 1144 | } |
| 1145 | } |
| 1146 | } |
| 1147 | // The tree the commit is to leave behind, which the stream describes as a |
| 1148 | // difference from the first parent. Building it explicitly is what lets a |
| 1149 | // merge be handled like anything else: whatever the operations of all the |
| 1150 | // parents render as, the commit ends by saying what git said it was. |
| 1151 | let mut target: BTreeMap<Vec<u8>, (Mode, Vec<u8>)> = BTreeMap::new(); |
| 1152 | for (path, file) in &state.at { |
| 1153 | let view = match state.repo.file(*file) { |
| 1154 | Some(v) => v, |
| 1155 | None => return Err(err!( |
| 1156 | "The file {} sits at {:?} and the render does not hold it.", |
| 1157 | file, tree::shown(path); |
| 1158 | Bug, Missing)), |
| 1159 | }; |
| 1160 | target.insert(path.clone(), (view.mode(), view.bytes().to_vec())); |
| 1161 | } |
| 1162 | res!(self.union(repo, &mut state, others, replica, &mut chain)); |
| 1163 | let before = repo.log.len(); |
| 1164 | for change in &commit.changes { |
| 1165 | match change { |
| 1166 | FileChange::Modify { mode, data, path } => { |
| 1167 | // A submodule is another repository and a bare tree entry is |
| 1168 | // not content, so neither is a file this history can hold. |
| 1169 | // Skipping either would leave a hole in every tree from here |
| 1170 | // on and say nothing about it, which is the wrong kind of |
| 1171 | // nothing. |
| 1172 | let mode = match mode.as_op_mode() { |
| 1173 | Some(m) => m, |
| 1174 | None => return Err(err!( |
| 1175 | "The commit {:?} on {:?} records {:?} with mode {}, which \ |
| 1176 | is {}. Neither is a file this repository can hold, and \ |
| 1177 | importing the tree without it would leave a hole nobody \ |
| 1178 | was told about.", |
| 1179 | subject(&commit.message), commit.refname, tree::shown(path), |
| 1180 | mode, |
| 1181 | match mode { |
| 1182 | FileMode::Gitlink => "a submodule", |
| 1183 | _ => "a directory entry", |
| 1184 | }; |
| 1185 | Invalid, Input, Unimplemented)), |
| 1186 | }; |
| 1187 | target.insert(path.clone(), (mode, res!(self.bytes_of(data)))); |
| 1188 | }, |
| 1189 | FileChange::Delete { path } => { |
| 1190 | target.remove(path); |
| 1191 | }, |
| 1192 | FileChange::Rename { src, dst } => { |
| 1193 | // A rename is the one change git states outright, so it is |
| 1194 | // recorded as one and the file keeps its identity through it. |
| 1195 | let held = match target.remove(src) { |
| 1196 | Some(b) => b, |
| 1197 | None => return Err(err!( |
| 1198 | "The commit on {:?} renames {:?}, which its parent does \ |
| 1199 | not hold.", commit.refname, tree::shown(src); |
| 1200 | Invalid, Input, Missing)), |
| 1201 | }; |
| 1202 | target.insert(dst.clone(), held); |
| 1203 | if let Some(file) = state.at.remove(src) { |
| 1204 | res!(self.mint(repo, &mut state, replica, &mut chain, |
| 1205 | Op::FileRename { file, path: dst.clone() })); |
| 1206 | state.at.insert(dst.clone(), file); |
| 1207 | } |
| 1208 | }, |
| 1209 | FileChange::Copy { src, dst } => { |
| 1210 | // The vocabulary has no copy, so the destination becomes a new |
| 1211 | // file holding the source's bytes -- and its mode, a copy in |
| 1212 | // git being of the tree entry and not of the blob alone. |
| 1213 | let held = match target.get(src) { |
| 1214 | Some(b) => b.clone(), |
| 1215 | None => return Err(err!( |
| 1216 | "The commit on {:?} copies {:?}, which its parent does \ |
| 1217 | not hold.", commit.refname, tree::shown(src); |
| 1218 | Invalid, Input, Missing)), |
| 1219 | }; |
| 1220 | target.insert(dst.clone(), held); |
| 1221 | }, |
| 1222 | FileChange::DeleteAll => target.clear(), |
| 1223 | // A note annotates a commit and says nothing about the tree. |
| 1224 | FileChange::Note { .. } => (), |
| 1225 | } |
| 1226 | } |
| 1227 | // Everything either side knows about, so that a file the merge dropped is |
| 1228 | // noticed as surely as one it wrote. |
| 1229 | let mut paths: Vec<Vec<u8>> = target.keys().cloned().collect(); |
| 1230 | for path in state.at.keys() { |
| 1231 | paths.push(path.clone()); |
| 1232 | } |
| 1233 | paths.sort(); |
| 1234 | paths.dedup(); |
| 1235 | for path in paths { |
| 1236 | // Taken rather than read, so that the bytes go into the splice that wants |
| 1237 | // them instead of being copied beside it. Nothing reads the tree again |
| 1238 | // once this loop has walked it. |
| 1239 | match target.remove(&path) { |
| 1240 | Some((mode, bytes)) => { |
| 1241 | res!(self.put( |
| 1242 | repo, &mut state, &mut chain, replica, &path, mode, &bytes)); |
| 1243 | }, |
| 1244 | None => { |
| 1245 | if let Some(file) = state.at.remove(&path) { |
| 1246 | res!(self.mint(repo, &mut state, replica, &mut chain, |
| 1247 | Op::FileDelete { file })); |
| 1248 | } |
| 1249 | }, |
| 1250 | } |
| 1251 | } |
| 1252 | if repo.log.len() != before { |
| 1253 | state.repo = Render::default(); |
| 1254 | state.repo = res!(self.render(&state.seq)); |
| 1255 | state.at = live_at(&state.repo); |
| 1256 | state.modes = modes_of(&state.repo); |
| 1257 | } |
| 1258 | let name = subject(&commit.message); |
| 1259 | // The author line goes into the body, where the mirror looks for it, so |
| 1260 | // that one person who wrote under three of them is one replica here and |
| 1261 | // still three names in a git mirror. See `op::AUTHOR_TRAILER`, which is |
| 1262 | // upstream because a mirror and a forge read what this writes. |
| 1263 | let said = Some(with_author( |
| 1264 | body(&commit.message).as_deref(), author_line(by).as_bytes())); |
| 1265 | // A date before the epoch, which git permits and which no history worth |
| 1266 | // importing holds, records as the epoch rather than as nothing: nothing |
| 1267 | // would send the mirror back to reusing the instant it first saw the |
| 1268 | // mark, and 1970 is nearer the truth than today is. |
| 1269 | let when = Some(by.when.secs.max(0) as u64); |
| 1270 | let head = res!(self.mint(repo, &mut state, replica, &mut chain, |
| 1271 | Op::Mark { name, body: said, time: when })); |
| 1272 | let mark = match commit.mark { |
| 1273 | Some(m) => m, |
| 1274 | None => { |
| 1275 | // A stream that marks nothing still needs each commit named, so |
| 1276 | // that a later one can point at it. |
| 1277 | self.minted += 1; |
| 1278 | u64::MAX - self.minted |
| 1279 | }, |
| 1280 | }; |
| 1281 | self.refs.insert(commit.refname.clone(), mark); |
| 1282 | self.commits.insert(mark, Done { head, state: Some(state) }); |
| 1283 | self.count += 1; |
| 1284 | // A commit is written to whichever reference names it, and where a |
| 1285 | // lightweight tag is that reference the tag's name is said here and |
| 1286 | // nowhere else in the stream. The commit's own mark is named after its |
| 1287 | // message, so the tag still needs one of its own. |
| 1288 | if let Some(name) = tag_name(&commit.refname) { |
| 1289 | let name = fmt!("{}", name); |
| 1290 | res!(self.lightweight_tag(repo, &name, mark)); |
| 1291 | } |
| 1292 | Ok(()) |
| 1293 | } |
| 1294 | |
| 1295 | /// Takes one event of the stream. |
| 1296 | fn event(&mut self, repo: &mut Repo, event: Event) |
| 1297 | -> Outcome<()> |
| 1298 | { |
| 1299 | match event { |
| 1300 | Event::Blob(blob) => { |
| 1301 | // The census resolves marks and parents and reads no content, so a |
| 1302 | // counting pass drops the bytes where they arrive. A blob no change |
| 1303 | // names -- a tag of a blob, a note the import skips -- is dropped in |
| 1304 | // the translating pass too, rather than held to the end for nobody. |
| 1305 | if let Some(mark) = blob.mark { |
| 1306 | if self.counting { |
| 1307 | // Read once, where the bytes are, and dropped with them. A |
| 1308 | // finding is a path, a line and a kind, so keeping those |
| 1309 | // costs nothing next to keeping the blob they came from. |
| 1310 | let found = secret::scan(&blob.data); |
| 1311 | if !found.is_empty() { |
| 1312 | self.found.insert(mark, found); |
| 1313 | } |
| 1314 | } else if self.wanted.contains_key(&mark) { |
| 1315 | self.blobs.insert(mark, blob.data); |
| 1316 | } |
| 1317 | } |
| 1318 | }, |
| 1319 | Event::Commit(commit) => res!(self.commit(repo, &commit)), |
| 1320 | Event::Reset { refname, from } => { |
| 1321 | if let Some(ObjRef::Mark(m)) = from { |
| 1322 | let at = self.resolve(m); |
| 1323 | self.refs.insert(refname.clone(), at); |
| 1324 | // A lightweight tag is a reference under refs/tags/ and |
| 1325 | // nothing more, and a mark is what says "this point in |
| 1326 | // history, by this name". So it imports as one. |
| 1327 | if let Some(name) = tag_name(&refname) { |
| 1328 | if !self.counting { |
| 1329 | res!(self.lightweight_tag(repo, name, at)); |
| 1330 | } |
| 1331 | } |
| 1332 | } |
| 1333 | }, |
| 1334 | Event::Alias { mark, to } => { |
| 1335 | if let ObjRef::Mark(m) = to { |
| 1336 | self.aliases.insert(mark, m); |
| 1337 | } |
| 1338 | }, |
| 1339 | // An annotated tag is an object of its own, carrying a message, a |
| 1340 | // tagger and possibly a signature. A mark carries a name and its place |
| 1341 | // in the history and nothing else, so flattening one to a mark would |
| 1342 | // drop a message the mirror could never write back out. |
| 1343 | Event::Tag(tag) => return Err(err!( |
| 1344 | "The repository holds the annotated tag {:?}, whose message and \ |
| 1345 | tagger no operation in this vocabulary can carry. A lightweight tag \ |
| 1346 | imports as a mark; an annotated one is refused rather than quietly \ |
| 1347 | flattened into one.", tag.name; |
| 1348 | Invalid, Input, Unimplemented)), |
| 1349 | // A note of progress, a checkpoint, a declared feature or an option |
| 1350 | // says nothing about the content of the tree. |
| 1351 | Event::Progress(_) | |
| 1352 | Event::Checkpoint | |
| 1353 | Event::Feature { .. } | |
| 1354 | Event::Opt(_) | |
| 1355 | Event::Done => (), |
| 1356 | } |
| 1357 | Ok(()) |
| 1358 | } |
| 1359 | } |
| 1360 | |
| 1361 | |
| 1362 | /// `ore import` -- reads a git repository, history and all. |
| 1363 | pub fn import(repo: &mut Repo, from: &Path) |
| 1364 | -> Outcome<()> |
| 1365 | { |
| 1366 | if !repo.log.is_empty() { |
| 1367 | return Err(err!( |
| 1368 | "This repository already holds {} operations; an import writes a \ |
| 1369 | history and will not be laid over one.", repo.log.len(); |
| 1370 | Invalid, Input, Conflict)); |
| 1371 | } |
| 1372 | let from = match std::fs::canonicalize(from) { |
| 1373 | Ok(p) => p, |
| 1374 | Err(e) => return Err(err!(e, |
| 1375 | "The git repository {:?} could not be resolved.", from; |
| 1376 | IO, File, Read)), |
| 1377 | }; |
| 1378 | res!(git(&from, &["rev-parse", "--git-dir"])); |
| 1379 | // Before the first operation, because there is no unwinding one. |
| 1380 | res!(refuse_tip_credentials(&from)); |
| 1381 | let branch = res!(git(&from, &["symbolic-ref", "--short", "HEAD"])); |
| 1382 | let refname = fmt!("refs/heads/{}", branch); |
| 1383 | |
| 1384 | // The stream is read twice: once to count which commits are named as parents, |
| 1385 | // and once to translate. The arguments are identical both times, so the marks |
| 1386 | // are identical both times, which is what makes the first pass describe the |
| 1387 | // second. Regenerating the stream costs git a few seconds; keeping every |
| 1388 | // commit's state instead cost gigabytes. |
| 1389 | let mut census = Import { counting: true, ..Default::default() }; |
| 1390 | res!(walk(&from, repo, &mut census)); |
| 1391 | // The census carries the findings forward with the counts, since it is the |
| 1392 | // pass that read the bytes and the translating pass never sees them whole. |
| 1393 | let carried = std::mem::take(&mut census.carried); |
| 1394 | let mut state = Import { |
| 1395 | needed: census.needed, |
| 1396 | wanted: census.wanted, |
| 1397 | ..Default::default() |
| 1398 | }; |
| 1399 | res!(walk(&from, repo, &mut state)); |
| 1400 | res!(repo.save_config()); |
| 1401 | println!("imported {} commit{} from {}", |
| 1402 | state.count, |
| 1403 | if state.count == 1 { "" } else { "s" }, |
| 1404 | from.display(), |
| 1405 | ); |
| 1406 | // Authors are counted as replicas rather than as identity lines, because the |
| 1407 | // mapping exists precisely so that several lines can be one person, and a |
| 1408 | // count that ignored it would report the thing the mapping was written to |
| 1409 | // prevent. |
| 1410 | let replicas: BTreeSet<u64> = repo.cfg.authors.values().copied().collect(); |
| 1411 | println!("{} operations, {} author{}", |
| 1412 | repo.log.len(), |
| 1413 | replicas.len(), |
| 1414 | if replicas.len() == 1 { "" } else { "s" }, |
| 1415 | ); |
| 1416 | if repo.cfg.authors.len() > replicas.len() { |
| 1417 | println!("{} identity line{} mapped onto them, as {:?} says", |
| 1418 | repo.cfg.authors.len(), |
| 1419 | if repo.cfg.authors.len() == 1 { "" } else { "s" }, |
| 1420 | repo.dir.join(CONFIG_FILE).display(), |
| 1421 | ); |
| 1422 | } |
| 1423 | if state.resolved > 0 { |
| 1424 | println!("{} file{} deleted where a merge brought two files to one path", |
| 1425 | state.resolved, if state.resolved == 1 { "" } else { "s" }); |
| 1426 | } |
| 1427 | if state.modes > 0 { |
| 1428 | println!("{} file mode{} recorded", state.modes, |
| 1429 | if state.modes == 1 { "" } else { "s" }); |
| 1430 | } |
| 1431 | if state.tags > 0 { |
| 1432 | println!("{} lightweight tag{} recorded as marks", state.tags, |
| 1433 | if state.tags == 1 { "" } else { "s" }); |
| 1434 | } |
| 1435 | // The detail goes to stderr and the count goes here, because a person who |
| 1436 | // redirected stdout to a file and never read stderr would otherwise have no |
| 1437 | // sign at all that this happened -- which is the shape of the failure that |
| 1438 | // `git fast-import` refusing a stream on stderr alone already taught. |
| 1439 | if let Some(summary) = report_carried(&carried) { |
| 1440 | println!("{}", summary); |
| 1441 | } |
| 1442 | for (who, replica) in &repo.cfg.authors { |
| 1443 | println!(" r{} {}", replica, who); |
| 1444 | } |
| 1445 | let mark = match state.refs.get(&refname) { |
| 1446 | Some(m) => *m, |
| 1447 | None => return Err(err!( |
| 1448 | "The stream carried no commit on {:?}, which is where HEAD points.", |
| 1449 | refname; |
| 1450 | Invalid, Input, Missing)), |
| 1451 | }; |
| 1452 | let head = match state.commits.get(&mark) { |
| 1453 | Some(d) => d.head, |
| 1454 | None => return Err(err!( |
| 1455 | "The commit :{} at the head of {:?} was not translated.", mark, refname; |
| 1456 | Bug, Missing)), |
| 1457 | }; |
| 1458 | let tree = res!(tree::whole(&repo.log)); |
| 1459 | let moved = res!(tree::materialise(&repo.root, &tree, tree::Surplus::Keep)); |
| 1460 | println!("the working copy is now the whole history; git had {} at {}", |
| 1461 | branch, head); |
| 1462 | println!(" {} created, {} updated, {} removed, {} left alone", |
| 1463 | moved.created.len(), |
| 1464 | moved.updated.len(), |
| 1465 | moved.removed.len(), |
| 1466 | moved.unchanged, |
| 1467 | ); |
| 1468 | for clash in &moved.clashes { |
| 1469 | println!(" clash at {}: {} keeps the name", |
| 1470 | tree::shown(&clash.path), clash.kept); |
| 1471 | for (file, name) in &clash.moved { |
| 1472 | println!(" {} is written as {}", file, tree::shown(name)); |
| 1473 | } |
| 1474 | } |
| 1475 | Ok(()) |
| 1476 | } |
| 1477 | |
| 1478 | /// Reads one whole `git fast-export` stream into whatever the caller is |
| 1479 | /// building, which is either the census or the import itself. |
| 1480 | fn walk(from: &Path, repo: &mut Repo, state: &mut Import) |
| 1481 | -> Outcome<()> |
| 1482 | { |
| 1483 | let mut child = match Command::new("git") |
| 1484 | .current_dir(from) |
| 1485 | .args(["fast-export", "--all", "-M"]) |
| 1486 | .stdout(Stdio::piped()) |
| 1487 | .stderr(Stdio::piped()) |
| 1488 | .spawn() |
| 1489 | { |
| 1490 | Ok(c) => c, |
| 1491 | Err(e) => return Err(err!(e, |
| 1492 | "`git fast-export` could not be started in {:?}.", from; |
| 1493 | IO, System)), |
| 1494 | }; |
| 1495 | let mut out = match child.stdout.take() { |
| 1496 | Some(o) => o, |
| 1497 | None => return Err(err!( |
| 1498 | "`git fast-export` gave no standard output to read."; |
| 1499 | IO, System, Missing)), |
| 1500 | }; |
| 1501 | let mut parser = Parser::new(); |
| 1502 | let mut buf = vec![0u8; CHUNK]; |
| 1503 | loop { |
| 1504 | let n = match out.read(&mut buf) { |
| 1505 | Ok(n) => n, |
| 1506 | Err(e) => return Err(err!(e, |
| 1507 | "The `git fast-export` stream could not be read."; |
| 1508 | IO, Read)), |
| 1509 | }; |
| 1510 | if n == 0 { |
| 1511 | break; |
| 1512 | } |
| 1513 | parser.feed(&buf[..n]); |
| 1514 | while let Some(event) = res!(parser.next_event()) { |
| 1515 | res!(state.event(repo, event)); |
| 1516 | } |
| 1517 | } |
| 1518 | parser.end(); |
| 1519 | while let Some(event) = res!(parser.next_event()) { |
| 1520 | res!(state.event(repo, event)); |
| 1521 | } |
| 1522 | let status = match child.wait() { |
| 1523 | Ok(s) => s, |
| 1524 | Err(e) => return Err(err!(e, |
| 1525 | "`git fast-export` could not be waited for."; |
| 1526 | IO, System)), |
| 1527 | }; |
| 1528 | if !status.success() { |
| 1529 | let mut text = String::new(); |
| 1530 | if let Some(mut err) = child.stderr.take() { |
| 1531 | let mut raw = Vec::new(); |
| 1532 | match err.read_to_end(&mut raw) { |
| 1533 | Ok(_) => text = fmt!("{}", String::from_utf8_lossy(&raw)), |
| 1534 | Err(_) => (), |
| 1535 | } |
| 1536 | } |
| 1537 | return Err(err!( |
| 1538 | "`git fast-export` failed in {:?}: {}", from, text.trim(); |
| 1539 | IO, System)); |
| 1540 | } |
| 1541 | Ok(()) |
| 1542 | } |