Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_file/src/office/docx/edit.rs

6.3 KiB, 5 runs

created by r1870400018:22925, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Editing a `.docx` in place: the runs that were asked to change, and not one byte more.
2//!
3//! # What survives, and why that is the requirement rather than a nicety
4//!
5//! The document being edited is somebody else's. Round-tripping it through
6//! [`oxedyne_fe2o3_text::doc`] -- read to Markdown, edit the Markdown, write a fresh `.docx` -- would
7//! produce a file that opens, looks about right, and has silently lost the comments, the bookmarks, the
8//! tracked changes, the content controls, the custom XML, the theme, the tab stops, the section
9//! properties and the headers. The person who finds out is not the user; it is the colleague they sent
10//! it to.
11//!
12//! So this changes `word/document.xml` by splicing bytes into the `<w:t>` elements that held the text
13//! being replaced, and [`crate::zip`] copies every other member of the archive verbatim. A document
14//! edited here differs from the one that arrived in exactly the runs that were edited.
15//!
16//! # Only the body, and it says so
17//!
18//! Headers, footers, footnotes, comments and text boxes outside the body are separate parts and are NOT
19//! searched. A phrase in a header therefore reports as absent rather than being quietly changed in one
20//! of two places -- and an absence that names the string is a caller's cue to look, where a partial
21//! replacement is a document that disagrees with itself.
22//!
23//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
24//! Anthropic Claude
25
26use crate::office::edit::{
27 Find,
28 Piece,
29 Tally,
30 apply,
31};
32use crate::office::opc;
33use crate::zip::{
34 Method,
35 Zip,
36};
37
38use oxedyne_fe2o3_core::prelude::*;
39use oxedyne_fe2o3_text::xml::{
40 Elem,
41 Xml,
42};
43use oxedyne_fe2o3_text::xml::write::escape;
44
45/// What an edit of a `.docx` produced.
46#[derive(Clone, Debug, Default)]
47pub struct Edited {
48 pub bytes: Vec<u8>,
49 pub tallies: Vec<Tally>, // one per edit asked for, in order
50 // Larger than the number of replacements wherever a phrase was spread across several runs, which
51 // is the ordinary case in a document a person has formatted.
52 pub runs: usize,
53}
54
55/// Replaces text in a `.docx`, leaving everything else exactly as it arrived.
56///
57/// An edit whose `find` is nowhere in the body is refused by name and NOTHING is written -- see
58/// [`crate::office::edit::apply`] on why a silent no-op is the failure this is written against.
59pub fn edit(bytes: &[u8], edits: &[Find]) -> Outcome<Edited> {
60 if edits.is_empty() {
61 return Err(err!("An edit of a document was asked for with no edits in it."; Invalid, Input));
62 }
63 let mut zip = res!(Zip::read(bytes.to_vec()));
64 let part = res!(opc::main_part(&zip, super::read::MAX_PART));
65 let src = res!(String::from_utf8(res!(zip.content_capped(&part, super::read::MAX_PART))),
66 Decode, String);
67 let mut xml = res!(Xml::parse(&src));
68 let groups = res!(paragraphs(&xml));
69 let (changes, tallies) = res!(apply(&groups, edits));
70 let runs = changes.len();
71 for c in &changes {
72 res!(xml.splice(c.piece.span.clone(), run_text(&c.text)));
73 }
74 zip.set(&part, xml.render().into_bytes(), Method::Deflate);
75 Ok(Edited { bytes: res!(zip.write()), tallies, runs })
76}
77
78/// The `<w:t>` elements of each paragraph, in document order, one group per paragraph.
79///
80/// A paragraph is the unit a match may not cross, because a sentence never spans one and a find that
81/// could would happily replace across a heading boundary.
82///
83/// A nested paragraph -- one inside a text box, which is inside a run, which is inside a paragraph --
84/// gets a group of its own and its runs are not also in the enclosing one. Counted twice they would
85/// produce two splices over the same bytes, which the splicer refuses; and it would be right to.
86fn paragraphs(xml: &Xml) -> Outcome<Vec<Vec<Piece>>> {
87 let mut out = Vec::new();
88 let root = res!(xml.root());
89 walk(xml, root, &mut out);
90 Ok(out)
91}
92
93/// Adds every paragraph at or below an element to the groups.
94fn walk(xml: &Xml, at: &Elem, out: &mut Vec<Vec<Piece>>) {
95 if at.name.qname == "w:p" {
96 // The slot is claimed before the nested paragraphs are walked, so an enclosing paragraph comes
97 // before the text box inside it and `nth` counts in a stable order.
98 let slot = out.len();
99 out.push(Vec::new());
100 let mut group = Vec::new();
101 gather(xml, at, &mut group, out);
102 out[slot] = group;
103 return;
104 }
105 for kid in at.elems() {
106 walk(xml, kid, out);
107 }
108}
109
110/// Adds one paragraph's own text runs to its group, handing a nested paragraph to [`walk`].
111fn gather(xml: &Xml, at: &Elem, group: &mut Vec<Piece>, out: &mut Vec<Vec<Piece>>) {
112 for kid in at.elems() {
113 match kid.name.qname.as_str() {
114 "w:p" => walk(xml, kid, out),
115 // `w:t` holds text a reader sees. `w:instrText` holds a field's INSTRUCTIONS -- a page
116 // reference, a merge field -- and replacing text in one changes what the field does.
117 "w:t" => group.push(Piece::new(kid.span.clone(), xml.text_of(kid))),
118 _ => gather(xml, kid, group, out),
119 }
120 }
121}
122
123/// A `<w:t>` holding this text.
124///
125/// `xml:space="preserve"` goes on wherever the text has whitespace at an end, and that is not
126/// optional: without it Word and every other reader collapse it, so replacing `Q1` with ` Q1 ` in a
127/// document that did not already carry the attribute writes a file whose text differs from what the
128/// edit asked for.
129fn run_text(text: &str) -> String {
130 let ends = text.starts_with(|c: char| c.is_whitespace())
131 || text.ends_with(|c: char| c.is_whitespace());
132 match ends {
133 true => fmt!("<w:t xml:space=\"preserve\">{}</w:t>", escape(text)),
134 false => fmt!("<w:t>{}</w:t>", escape(text)),
135 }
136}
137
138/// The text of the body, run by run and paragraph by paragraph, as an edit sees it.
139///
140/// What a caller shows a person before asking them to confirm an edit, and what a test asserts against.
141/// It is NOT the document as prose -- there is [`super::read`] for that -- it is the strings a `find`
142/// is matched against, which is a different thing wherever a writer split a sentence.
143pub fn body_text(bytes: &[u8]) -> Outcome<Vec<String>> {
144 let zip = res!(Zip::read(bytes.to_vec()));
145 let part = res!(opc::main_part(&zip, super::read::MAX_PART));
146 let src = res!(String::from_utf8(res!(zip.content_capped(&part, super::read::MAX_PART))),
147 Decode, String);
148 let xml = res!(Xml::parse(&src));
149 let groups = res!(paragraphs(&xml));
150 Ok(groups.iter()
151 .map(|g| g.iter().map(|p| p.text.as_str()).collect::<Vec<_>>().concat())
152 .collect())
153}