oxedyne/fe2o3/fe2o3_file/src/office/docx/edit.rs
6.3 KiB, 5 runs
created by r1870400018:22925, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Editing a `.docx` in place: the runs that were asked to change, and not one byte more. |
| 2 | //! |
| 3 | //! # What survives, and why that is the requirement rather than a nicety |
| 4 | //! |
| 5 | //! The document being edited is somebody else's. Round-tripping it through |
| 6 | //! [`oxedyne_fe2o3_text::doc`] -- read to Markdown, edit the Markdown, write a fresh `.docx` -- would |
| 7 | //! produce a file that opens, looks about right, and has silently lost the comments, the bookmarks, the |
| 8 | //! tracked changes, the content controls, the custom XML, the theme, the tab stops, the section |
| 9 | //! properties and the headers. The person who finds out is not the user; it is the colleague they sent |
| 10 | //! it to. |
| 11 | //! |
| 12 | //! So this changes `word/document.xml` by splicing bytes into the `<w:t>` elements that held the text |
| 13 | //! being replaced, and [`crate::zip`] copies every other member of the archive verbatim. A document |
| 14 | //! edited here differs from the one that arrived in exactly the runs that were edited. |
| 15 | //! |
| 16 | //! # Only the body, and it says so |
| 17 | //! |
| 18 | //! Headers, footers, footnotes, comments and text boxes outside the body are separate parts and are NOT |
| 19 | //! searched. A phrase in a header therefore reports as absent rather than being quietly changed in one |
| 20 | //! of two places -- and an absence that names the string is a caller's cue to look, where a partial |
| 21 | //! replacement is a document that disagrees with itself. |
| 22 | //! |
| 23 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 24 | //! Anthropic Claude |
| 25 | |
| 26 | use crate::office::edit::{ |
| 27 | Find, |
| 28 | Piece, |
| 29 | Tally, |
| 30 | apply, |
| 31 | }; |
| 32 | use crate::office::opc; |
| 33 | use crate::zip::{ |
| 34 | Method, |
| 35 | Zip, |
| 36 | }; |
| 37 | |
| 38 | use oxedyne_fe2o3_core::prelude::*; |
| 39 | use oxedyne_fe2o3_text::xml::{ |
| 40 | Elem, |
| 41 | Xml, |
| 42 | }; |
| 43 | use oxedyne_fe2o3_text::xml::write::escape; |
| 44 | |
| 45 | /// What an edit of a `.docx` produced. |
| 46 | #[derive(Clone, Debug, Default)] |
| 47 | pub struct Edited { |
| 48 | pub bytes: Vec<u8>, |
| 49 | pub tallies: Vec<Tally>, // one per edit asked for, in order |
| 50 | // Larger than the number of replacements wherever a phrase was spread across several runs, which |
| 51 | // is the ordinary case in a document a person has formatted. |
| 52 | pub runs: usize, |
| 53 | } |
| 54 | |
| 55 | /// Replaces text in a `.docx`, leaving everything else exactly as it arrived. |
| 56 | /// |
| 57 | /// An edit whose `find` is nowhere in the body is refused by name and NOTHING is written -- see |
| 58 | /// [`crate::office::edit::apply`] on why a silent no-op is the failure this is written against. |
| 59 | pub fn edit(bytes: &[u8], edits: &[Find]) -> Outcome<Edited> { |
| 60 | if edits.is_empty() { |
| 61 | return Err(err!("An edit of a document was asked for with no edits in it."; Invalid, Input)); |
| 62 | } |
| 63 | let mut zip = res!(Zip::read(bytes.to_vec())); |
| 64 | let part = res!(opc::main_part(&zip, super::read::MAX_PART)); |
| 65 | let src = res!(String::from_utf8(res!(zip.content_capped(&part, super::read::MAX_PART))), |
| 66 | Decode, String); |
| 67 | let mut xml = res!(Xml::parse(&src)); |
| 68 | let groups = res!(paragraphs(&xml)); |
| 69 | let (changes, tallies) = res!(apply(&groups, edits)); |
| 70 | let runs = changes.len(); |
| 71 | for c in &changes { |
| 72 | res!(xml.splice(c.piece.span.clone(), run_text(&c.text))); |
| 73 | } |
| 74 | zip.set(&part, xml.render().into_bytes(), Method::Deflate); |
| 75 | Ok(Edited { bytes: res!(zip.write()), tallies, runs }) |
| 76 | } |
| 77 | |
| 78 | /// The `<w:t>` elements of each paragraph, in document order, one group per paragraph. |
| 79 | /// |
| 80 | /// A paragraph is the unit a match may not cross, because a sentence never spans one and a find that |
| 81 | /// could would happily replace across a heading boundary. |
| 82 | /// |
| 83 | /// A nested paragraph -- one inside a text box, which is inside a run, which is inside a paragraph -- |
| 84 | /// gets a group of its own and its runs are not also in the enclosing one. Counted twice they would |
| 85 | /// produce two splices over the same bytes, which the splicer refuses; and it would be right to. |
| 86 | fn paragraphs(xml: &Xml) -> Outcome<Vec<Vec<Piece>>> { |
| 87 | let mut out = Vec::new(); |
| 88 | let root = res!(xml.root()); |
| 89 | walk(xml, root, &mut out); |
| 90 | Ok(out) |
| 91 | } |
| 92 | |
| 93 | /// Adds every paragraph at or below an element to the groups. |
| 94 | fn walk(xml: &Xml, at: &Elem, out: &mut Vec<Vec<Piece>>) { |
| 95 | if at.name.qname == "w:p" { |
| 96 | // The slot is claimed before the nested paragraphs are walked, so an enclosing paragraph comes |
| 97 | // before the text box inside it and `nth` counts in a stable order. |
| 98 | let slot = out.len(); |
| 99 | out.push(Vec::new()); |
| 100 | let mut group = Vec::new(); |
| 101 | gather(xml, at, &mut group, out); |
| 102 | out[slot] = group; |
| 103 | return; |
| 104 | } |
| 105 | for kid in at.elems() { |
| 106 | walk(xml, kid, out); |
| 107 | } |
| 108 | } |
| 109 | |
| 110 | /// Adds one paragraph's own text runs to its group, handing a nested paragraph to [`walk`]. |
| 111 | fn gather(xml: &Xml, at: &Elem, group: &mut Vec<Piece>, out: &mut Vec<Vec<Piece>>) { |
| 112 | for kid in at.elems() { |
| 113 | match kid.name.qname.as_str() { |
| 114 | "w:p" => walk(xml, kid, out), |
| 115 | // `w:t` holds text a reader sees. `w:instrText` holds a field's INSTRUCTIONS -- a page |
| 116 | // reference, a merge field -- and replacing text in one changes what the field does. |
| 117 | "w:t" => group.push(Piece::new(kid.span.clone(), xml.text_of(kid))), |
| 118 | _ => gather(xml, kid, group, out), |
| 119 | } |
| 120 | } |
| 121 | } |
| 122 | |
| 123 | /// A `<w:t>` holding this text. |
| 124 | /// |
| 125 | /// `xml:space="preserve"` goes on wherever the text has whitespace at an end, and that is not |
| 126 | /// optional: without it Word and every other reader collapse it, so replacing `Q1` with ` Q1 ` in a |
| 127 | /// document that did not already carry the attribute writes a file whose text differs from what the |
| 128 | /// edit asked for. |
| 129 | fn run_text(text: &str) -> String { |
| 130 | let ends = text.starts_with(|c: char| c.is_whitespace()) |
| 131 | || text.ends_with(|c: char| c.is_whitespace()); |
| 132 | match ends { |
| 133 | true => fmt!("<w:t xml:space=\"preserve\">{}</w:t>", escape(text)), |
| 134 | false => fmt!("<w:t>{}</w:t>", escape(text)), |
| 135 | } |
| 136 | } |
| 137 | |
| 138 | /// The text of the body, run by run and paragraph by paragraph, as an edit sees it. |
| 139 | /// |
| 140 | /// What a caller shows a person before asking them to confirm an edit, and what a test asserts against. |
| 141 | /// It is NOT the document as prose -- there is [`super::read`] for that -- it is the strings a `find` |
| 142 | /// is matched against, which is a different thing wherever a writer split a sentence. |
| 143 | pub fn body_text(bytes: &[u8]) -> Outcome<Vec<String>> { |
| 144 | let zip = res!(Zip::read(bytes.to_vec())); |
| 145 | let part = res!(opc::main_part(&zip, super::read::MAX_PART)); |
| 146 | let src = res!(String::from_utf8(res!(zip.content_capped(&part, super::read::MAX_PART))), |
| 147 | Decode, String); |
| 148 | let xml = res!(Xml::parse(&src)); |
| 149 | let groups = res!(paragraphs(&xml)); |
| 150 | Ok(groups.iter() |
| 151 | .map(|g| g.iter().map(|p| p.text.as_str()).collect::<Vec<_>>().concat()) |
| 152 | .collect()) |
| 153 | } |