oxedyne/fe2o3/fe2o3_graphics/src/pdf.rs
75.8 KiB, 281 runs
created by r1870400018:36022, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! A minimal PDF writer: pages of filled and stroked outline paths, and text in embedded fonts. |
| 2 | //! |
| 3 | //! Text reaches the page in one of two ways. A glyph from an embeddable font file ([`FontProgram`]) is |
| 4 | //! shown by `TJ` from a `Type0` font over a subset of that very file, so a viewer renders the designer's |
| 5 | //! outlines with its own rasteriser and copies, searches and reads aloud the text through the font's |
| 6 | //! `/ToUnicode`. A glyph whose font cannot be embedded -- a variable `CFF2` face, or a licence that |
| 7 | //! forbids it -- falls back to its filled outline, stored once in a Type-3 font that carries a |
| 8 | //! `/ToUnicode` of its own, so it too stays extractable. |
| 9 | //! |
| 10 | //! The geometry and colour are the crate's own [`Path`], [`Pt`], [`Seg`] and [`Rgba`]; nothing here |
| 11 | //! defines a parallel type. A quadratic segment is elevated to a cubic on the way out, since PDF has |
| 12 | //! no quadratic operator, and the whole page is flipped in y so the engine's top-left, y-down frame |
| 13 | //! meets PDF's bottom-left, y-up one. |
| 14 | //! |
| 15 | //! The bytes are deterministic: no dates are written, the `/ID` is derived from the file's own |
| 16 | //! content rather than the clock, and no producer string leaks a version or a build. The same page |
| 17 | //! list yields the same bytes on every run, which is what a content-addressed pipeline needs. |
| 18 | //! |
| 19 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 20 | //! Anthropic Claude |
| 21 | |
| 22 | use crate::colour::Rgba; |
| 23 | use crate::pdf_font::{ |
| 24 | FontFile, |
| 25 | FontProgram, |
| 26 | Outlines, |
| 27 | }; |
| 28 | use crate::path::{ |
| 29 | Path, |
| 30 | Pt, |
| 31 | Seg, |
| 32 | }; |
| 33 | use crate::transform::Transform; |
| 34 | |
| 35 | use std::collections::BTreeMap; |
| 36 | use std::collections::BTreeSet; |
| 37 | use std::collections::HashMap; |
| 38 | use std::io::Write; |
| 39 | use std::sync::Arc; |
| 40 | |
| 41 | use oxedyne_fe2o3_core::prelude::*; |
| 42 | |
| 43 | // The FNV-1a parameters the file's deterministic `/ID` is folded with: two bases give the two hash |
| 44 | // halves, and the one prime multiplies each step. Kept as constants so the streaming writer can fold |
| 45 | // the body as it goes rather than hashing a whole buffer at the end. |
| 46 | const FNV_BASIS_A: u64 = 0xcbf2_9ce4_8422_2325; |
| 47 | const FNV_BASIS_B: u64 = 0x8422_2325_cbf2_9ce4; |
| 48 | const FNV_PRIME: u64 = 0x0000_0100_0000_01b3; |
| 49 | |
| 50 | // A glyph outline is stored once, in the font frame, at this many units per em, and shown at any point |
| 51 | // size by the Type-3 `/FontMatrix` and text-showing size. A single canonical size means the same glyph |
| 52 | // drawn at two sizes shares one stored outline. |
| 53 | const GLYPH_UPM: f32 = 1000.0; |
| 54 | |
| 55 | /// A 64-bit FNV-1a over bytes: the content key a glyph outline is deduplicated by. Two glyphs whose |
| 56 | /// path operators serialise identically share one Type-3 `CharProc`, whatever face or size drew them, |
| 57 | /// which is what a content key buys over a `(face, id)` pair that means nothing across font chains. |
| 58 | fn fnv1a(bytes: &[u8]) -> u64 { |
| 59 | let mut h = FNV_BASIS_A; |
| 60 | for &b in bytes { |
| 61 | h ^= b as u64; |
| 62 | h = h.wrapping_mul(FNV_PRIME); |
| 63 | } |
| 64 | h |
| 65 | } |
| 66 | |
| 67 | /// One drawn shape: a path and how it is painted, or a raster image placed in a rectangle. Fill and |
| 68 | /// stroke are the two the typesetter needs for vector ink -- glyphs and rules fill, a held-open |
| 69 | /// reservation strokes; `Image` embeds a decoded raster (a figure's photograph or diagram) as an image |
| 70 | /// XObject, its samples straight RGB with an optional grey soft mask for translucency. |
| 71 | #[derive(Clone, Debug)] |
| 72 | pub enum Draw { |
| 73 | Fill { |
| 74 | path: Path, |
| 75 | colour: Rgba, |
| 76 | }, |
| 77 | Stroke { |
| 78 | path: Path, |
| 79 | colour: Rgba, |
| 80 | width: f64, // pen width, in points |
| 81 | }, |
| 82 | Image { |
| 83 | rgb: Vec<u8>, // packed RGB, iw*ih*3, row-major, top row first |
| 84 | alpha: Option<Vec<u8>>, // packed grey soft mask, iw*ih, present only when a pixel is translucent |
| 85 | iw: usize, // image width in samples |
| 86 | ih: usize, // image height in samples |
| 87 | x: f64, // placement rectangle, engine frame (top-left, y down), points |
| 88 | y: f64, |
| 89 | w: f64, |
| 90 | h: f64, |
| 91 | }, |
| 92 | Glyph { |
| 93 | outline: Path, // the outline in the font frame, y up, at GLYPH_UPM units per em |
| 94 | x: f32, // pen x in the engine frame (top-left, y down), points |
| 95 | y: f32, // pen y (the baseline) in the engine frame, points |
| 96 | size: f32, // point size the glyph is shown at |
| 97 | adv: f32, // advance in points at that size, for the font's /Widths |
| 98 | colour: Rgba, |
| 99 | text: String, // source scalar(s) this glyph stands for, for /ToUnicode; may be empty |
| 100 | }, |
| 101 | Text { |
| 102 | font: Arc<FontProgram>, |
| 103 | gid: u16, // glyph id in the program, shown as the CID of the same number |
| 104 | x: f32, // pen x in the engine frame (top-left, y down), points |
| 105 | y: f32, // the baseline in the engine frame, points |
| 106 | size: f32, // point size |
| 107 | colour: Rgba, |
| 108 | text: String, // source scalar(s), for /ToUnicode; may be empty |
| 109 | }, |
| 110 | } |
| 111 | |
| 112 | impl Draw { |
| 113 | |
| 114 | /// The paint colour of a vector draw, opaque for an image (whose translucency rides its own soft |
| 115 | /// mask, not the page's alpha graphics state). |
| 116 | fn colour(&self) -> Rgba { |
| 117 | match self { |
| 118 | Draw::Fill { colour, .. } => *colour, |
| 119 | Draw::Stroke { colour, .. } => *colour, |
| 120 | Draw::Glyph { colour, .. } => *colour, |
| 121 | Draw::Text { colour, .. } => *colour, |
| 122 | Draw::Image { .. } => Rgba::new(0, 0, 0, 255), |
| 123 | } |
| 124 | } |
| 125 | } |
| 126 | |
| 127 | /// A clickable link over a rectangle of the page: a `/Link` annotation whose action opens a URI. The |
| 128 | /// rectangle is in the engine's page frame -- top-left origin, y down, in points -- exactly as a draw's |
| 129 | /// coordinates are, so a caller hands over the same rectangle it placed the ink in and [`PdfStream`] |
| 130 | /// flips it into PDF's y-up frame when it writes the annotation. The border is drawn away, so a linked |
| 131 | /// image reads as the picture alone. |
| 132 | #[derive(Clone, Debug)] |
| 133 | pub struct LinkAnnot { |
| 134 | pub x0: f64, // left, engine frame (points) |
| 135 | pub y0: f64, // top |
| 136 | pub x1: f64, // right |
| 137 | pub y1: f64, // bottom |
| 138 | pub uri: String, // the destination the link opens |
| 139 | } |
| 140 | |
| 141 | /// One page: its size in points, the shapes drawn on it back to front, and any clickable link |
| 142 | /// annotations over it. |
| 143 | /// |
| 144 | /// The coordinates in the paths are the engine's page frame -- top-left origin, y increasing |
| 145 | /// downwards -- exactly as the SVG writer receives them. [`PdfWriter`] applies the flip to PDF's |
| 146 | /// y-up frame itself, so a caller hands the same paths to either writer. |
| 147 | #[derive(Clone, Debug)] |
| 148 | pub struct PdfPage { |
| 149 | pub width: f64, // media box width, in points |
| 150 | pub height: f64, // media box height, in points |
| 151 | pub draws: Vec<Draw>, |
| 152 | pub annots: Vec<LinkAnnot>, // clickable link rectangles; a page with none writes no /Annots |
| 153 | } |
| 154 | |
| 155 | impl PdfPage { |
| 156 | |
| 157 | pub fn new(width: f64, height: f64) -> Self { |
| 158 | Self { width, height, draws: Vec::new(), annots: Vec::new() } |
| 159 | } |
| 160 | |
| 161 | /// Adds a clickable link over the rectangle at top-left `(x, y)`, `w` wide and `h` tall in the |
| 162 | /// engine's y-down point frame -- the same frame [`image`](Self::image) places a raster in -- opening |
| 163 | /// `uri`. The page gains an `/Annots` entry; a page with no link stays byte-identical to before. |
| 164 | pub fn link(&mut self, x: f64, y: f64, w: f64, h: f64, uri: String) { |
| 165 | self.annots.push(LinkAnnot { x0: x, y0: y, x1: x + w, y1: y + h, uri }); |
| 166 | } |
| 167 | |
| 168 | pub fn fill(&mut self, path: Path, colour: Rgba) { |
| 169 | self.draws.push(Draw::Fill { path, colour }); |
| 170 | } |
| 171 | |
| 172 | pub fn stroke(&mut self, path: Path, colour: Rgba, width: f64) { |
| 173 | self.draws.push(Draw::Stroke { path, colour, width }); |
| 174 | } |
| 175 | |
| 176 | /// Places a decoded raster in the rectangle at top-left `(x, y)`, `w` wide and `h` tall, in the |
| 177 | /// engine's y-down point frame. `rgb` is `iw * ih * 3` straight-RGB samples, top row first; `alpha`, |
| 178 | /// when given, is the matching `iw * ih` grey soft mask that carries any translucency. The image is |
| 179 | /// scaled to fill the rectangle, so the caller sizes the rectangle to the image's aspect if it wants |
| 180 | /// no distortion. |
| 181 | #[allow(clippy::too_many_arguments)] |
| 182 | pub fn image( |
| 183 | &mut self, |
| 184 | rgb: Vec<u8>, |
| 185 | alpha: Option<Vec<u8>>, |
| 186 | iw: usize, |
| 187 | ih: usize, |
| 188 | x: f64, |
| 189 | y: f64, |
| 190 | w: f64, |
| 191 | h: f64, |
| 192 | ) { |
| 193 | self.draws.push(Draw::Image { rgb, alpha, iw, ih, x, y, w, h }); |
| 194 | } |
| 195 | |
| 196 | /// Adds a glyph placed at pen `(x, y)` -- `x` the left, `y` the baseline, in the engine's top-left, |
| 197 | /// y-down point frame -- shown at `size` points and painted `colour`. `outline` is the glyph in the |
| 198 | /// font frame (y up, at [`GLYPH_UPM`] units per em); `adv` is its advance in points for the font's |
| 199 | /// `/Widths`; `text` is the source scalar(s) it stands for, recorded in the font's `/ToUnicode` so the |
| 200 | /// glyph is text-extractable, or empty when none is known. The writer stores each distinct outline |
| 201 | /// once as a Type-3 `CharProc` and references it here, so a glyph drawn a thousand times costs its |
| 202 | /// outline once. |
| 203 | pub fn glyph(&mut self, outline: Path, x: f32, y: f32, size: f32, adv: f32, colour: Rgba, text: String) { |
| 204 | self.draws.push(Draw::Glyph { outline, x, y, size, adv, colour, text }); |
| 205 | } |
| 206 | |
| 207 | /// Adds glyph `gid` of an embeddable font at pen `(x, y)` -- `x` the left, `y` the baseline, in the |
| 208 | /// engine's top-left, y-down point frame -- shown at `size` points and painted `colour`. `text` is the |
| 209 | /// source it stands for, recorded in the font's `/ToUnicode`; the first non-empty text a glyph is met |
| 210 | /// with is the one it keeps. The font is embedded once, subset to the glyphs the document shows. |
| 211 | #[allow(clippy::too_many_arguments)] |
| 212 | pub fn text( |
| 213 | &mut self, |
| 214 | font: Arc<FontProgram>, |
| 215 | gid: u16, |
| 216 | x: f32, |
| 217 | y: f32, |
| 218 | size: f32, |
| 219 | colour: Rgba, |
| 220 | text: String, |
| 221 | ) { |
| 222 | self.draws.push(Draw::Text { font, gid, x, y, size, colour, text }); |
| 223 | } |
| 224 | } |
| 225 | |
| 226 | /// One entry in the document outline (the viewer's bookmark side panel): a title, the zero-based page |
| 227 | /// it jumps to, and its nesting depth. Depth zero is a top-level entry; a deeper entry nests under the |
| 228 | /// nearest preceding entry of a shallower depth, so a flat list in reading order builds the tree. The |
| 229 | /// destination is the top of the page fitted to the window, which every entry shares -- the outline |
| 230 | /// names pages, not positions within them. |
| 231 | #[derive(Clone, Debug)] |
| 232 | pub struct OutlineItem { |
| 233 | pub title: String, |
| 234 | pub page: usize, // zero-based page index the entry jumps to |
| 235 | pub level: u8, // nesting depth, zero at the top |
| 236 | } |
| 237 | |
| 238 | /// The parent, sibling and child links one outline item needs, resolved from the flat level list. |
| 239 | /// Indices are into the item slice; `count` is the number of descendants, always shown open. |
| 240 | struct OutlineLinks { |
| 241 | parent: Option<usize>, |
| 242 | prev: Option<usize>, |
| 243 | next: Option<usize>, |
| 244 | first: Option<usize>, |
| 245 | last: Option<usize>, |
| 246 | count: usize, |
| 247 | } |
| 248 | |
| 249 | /// Resolves the flat, reading-order outline list into a tree: each item's parent, siblings and |
| 250 | /// children, and the roots. A stack of open ancestors gives the nearest shallower item as parent, so |
| 251 | /// a level that skips a depth still nests sensibly. |
| 252 | fn build_outline_links(items: &[OutlineItem]) -> (Vec<OutlineLinks>, Vec<usize>) { |
| 253 | let n = items.len(); |
| 254 | let mut children: Vec<Vec<usize>> = vec![Vec::new(); n]; |
| 255 | let mut parent: Vec<Option<usize>> = vec![None; n]; |
| 256 | let mut roots: Vec<usize> = Vec::new(); |
| 257 | let mut stack: Vec<usize> = Vec::new(); |
| 258 | for i in 0..n { |
| 259 | while let Some(&t) = stack.last() { |
| 260 | if items[t].level >= items[i].level { |
| 261 | stack.pop(); |
| 262 | } else { |
| 263 | break; |
| 264 | } |
| 265 | } |
| 266 | match stack.last() { |
| 267 | Some(&p) => { |
| 268 | parent[i] = Some(p); |
| 269 | children[p].push(i); |
| 270 | }, |
| 271 | None => roots.push(i), |
| 272 | } |
| 273 | stack.push(i); |
| 274 | } |
| 275 | |
| 276 | // The descendant count of a pre-order list is the run of following items whose level stays deeper. |
| 277 | let mut links = Vec::with_capacity(n); |
| 278 | for i in 0..n { |
| 279 | let siblings = match parent[i] { |
| 280 | Some(p) => &children[p], |
| 281 | None => &roots, |
| 282 | }; |
| 283 | let at = siblings.iter().position(|&j| j == i).unwrap_or(0); |
| 284 | let prev = if at > 0 { Some(siblings[at - 1]) } else { None }; |
| 285 | let next = siblings.get(at + 1).copied(); |
| 286 | let first = children[i].first().copied(); |
| 287 | let last = children[i].last().copied(); |
| 288 | let mut count = 0usize; |
| 289 | let mut j = i + 1; |
| 290 | while j < n && items[j].level > items[i].level { |
| 291 | count += 1; |
| 292 | j += 1; |
| 293 | } |
| 294 | links.push(OutlineLinks { parent: parent[i], prev, next, first, last, count }); |
| 295 | } |
| 296 | (links, roots) |
| 297 | } |
| 298 | |
| 299 | /// A PDF text string for a title: a parenthesised literal with `(`, `)` and `\` escaped when the text |
| 300 | /// is printable ASCII, else a UTF-16BE hex string with a byte-order mark so any Unicode renders. Both |
| 301 | /// forms are deterministic, which the content-addressed file needs. |
| 302 | fn pdf_text_string(s: &str) -> String { |
| 303 | if s.bytes().all(|b| (0x20..0x7f).contains(&b)) { |
| 304 | let mut out = String::from("("); |
| 305 | for c in s.chars() { |
| 306 | match c { |
| 307 | '(' => out.push_str("\\("), |
| 308 | ')' => out.push_str("\\)"), |
| 309 | '\\' => out.push_str("\\\\"), |
| 310 | _ => out.push(c), |
| 311 | } |
| 312 | } |
| 313 | out.push(')'); |
| 314 | out |
| 315 | } else { |
| 316 | let mut out = String::from("<FEFF"); |
| 317 | for u in s.encode_utf16() { |
| 318 | out.push_str(&fmt!("{:04X}", u)); |
| 319 | } |
| 320 | out.push('>'); |
| 321 | out |
| 322 | } |
| 323 | } |
| 324 | |
| 325 | /// Accumulates pages and writes them out as one PDF file. |
| 326 | #[derive(Clone, Debug, Default)] |
| 327 | pub struct PdfWriter { |
| 328 | pages: Vec<PdfPage>, |
| 329 | compress: bool, |
| 330 | outline: Vec<OutlineItem>, |
| 331 | } |
| 332 | |
| 333 | impl PdfWriter { |
| 334 | |
| 335 | pub fn new() -> Self { |
| 336 | Self::default() |
| 337 | } |
| 338 | |
| 339 | /// Compress each content stream with zlib and mark it `/FlateDecode`. Off by default: an |
| 340 | /// uncompressed stream is trivially deterministic and easy to read while the writer is young. |
| 341 | pub fn with_compression(mut self, on: bool) -> Self { |
| 342 | self.compress = on; |
| 343 | self |
| 344 | } |
| 345 | |
| 346 | pub fn add_page(&mut self, page: PdfPage) { |
| 347 | self.pages.push(page); |
| 348 | } |
| 349 | |
| 350 | /// Sets the document outline (the viewer's bookmark side panel), replacing any earlier one. An empty |
| 351 | /// list leaves the file with no outline, byte for byte as before the feature existed. |
| 352 | pub fn set_outline(&mut self, outline: Vec<OutlineItem>) { |
| 353 | self.outline = outline; |
| 354 | } |
| 355 | |
| 356 | /// Renders the whole document to PDF bytes. |
| 357 | /// |
| 358 | /// A convenience over [`PdfStream`]: it streams the accumulated pages into an in-memory buffer and |
| 359 | /// returns it. The bytes are exactly those [`PdfStream`] writes page by page, so a caller that |
| 360 | /// cannot hold the whole document keeps the identical file by streaming to a file handle instead. |
| 361 | pub fn to_bytes(&self) -> Outcome<Vec<u8>> { |
| 362 | let mut stream = res!(PdfStream::new_with_outline( |
| 363 | Vec::new(), self.pages.len(), self.compress, self.outline.clone())); |
| 364 | for page in &self.pages { |
| 365 | res!(stream.page(page)); |
| 366 | } |
| 367 | Ok(res!(stream.finish())) |
| 368 | } |
| 369 | } |
| 370 | |
| 371 | /// A PDF writer that emits one page at a time to any [`Write`] sink, holding no more than the page in |
| 372 | /// hand. Where [`PdfWriter`] accumulates every page's outline paths and then serialises them into one |
| 373 | /// buffer -- three live copies of the whole document at the peak -- this writes each page's objects |
| 374 | /// the moment it is handed over and lets the caller drop the page, so a book of any length costs one |
| 375 | /// page of memory. The bytes are identical to [`PdfWriter::to_bytes`]: the same object numbering, the |
| 376 | /// same body order, the same content-derived `/ID`. |
| 377 | /// |
| 378 | /// The page count is fixed at construction because the page-tree object -- written first, before any |
| 379 | /// page -- names every page object and their count. The deterministic `/ID` is a hash of the body, |
| 380 | /// folded here as each byte is written rather than over a finished buffer, so no buffer is needed. |
| 381 | /// |
| 382 | /// Text is written in fonts gathered as pages are written -- in page order, on the writer's thread, so |
| 383 | /// every number and code is assigned once and deterministically. An embedded font records the glyph ids |
| 384 | /// shown from it and is subset to them in [`finish`](Self::finish), when the set is complete; an outline |
| 385 | /// glyph is stored once as a Type-3 `CharProc` and shown by a one-byte code. Both kinds of font object |
| 386 | /// are written after the last page, on numbers reserved as each font is first needed. |
| 387 | pub struct PdfStream<W: Write> { |
| 388 | out: W, |
| 389 | compress: bool, |
| 390 | n: usize, // the fixed page count, named by the page tree |
| 391 | offsets: Vec<usize>, // one-based object byte offsets; [0] is the free object |
| 392 | pos: usize, // bytes of body written so far, the next object's offset |
| 393 | added: usize, // pages handed over so far |
| 394 | next_extra: usize, // next free object number past the fixed block and the outline, for image XObjects |
| 395 | hash_a: u64, // running FNV-1a of the body, first `/ID` half |
| 396 | hash_b: u64, // running FNV-1a of the body, second half |
| 397 | outline: Vec<OutlineItem>, // document outline entries, empty for none |
| 398 | outline_root: usize, // object number of the /Outlines dict, zero when there is no outline |
| 399 | glyph_slots: HashMap<u64, (usize, u8)>, // outline content key -> (font index, code) |
| 400 | fonts: Vec<Type3Font>, // one Type-3 font per 256 distinct glyphs, in assignment order |
| 401 | cid_slots: HashMap<u64, usize>, // font program key -> index into `cid_fonts` |
| 402 | cid_fonts: Vec<CidFont>, // one embedded font per program, in order of first use |
| 403 | } |
| 404 | |
| 405 | /// One embedded font: the program, the glyphs shown from it with the text each stands for, and the |
| 406 | /// object number reserved for its `Type0` dictionary when it was first needed. |
| 407 | struct CidFont { |
| 408 | obj: usize, |
| 409 | prog: Arc<FontProgram>, |
| 410 | used: BTreeMap<u16, String>, |
| 411 | } |
| 412 | |
| 413 | /// The text-object state of a content stream being built: whether a `BT` is open, the font, size and |
| 414 | /// colour last set in it, and the `TJ` run being gathered. |
| 415 | #[derive(Default)] |
| 416 | struct TextState { |
| 417 | open: bool, |
| 418 | font: Option<(String, f32)>, |
| 419 | colour: Option<Rgba>, |
| 420 | run: Option<TjRun>, |
| 421 | } |
| 422 | |
| 423 | /// Consecutive embedded glyphs on one baseline, shown by a single `TJ` whose numbers carry the |
| 424 | /// difference between each glyph's placed position and where the font's own advance leaves the pen. |
| 425 | struct TjRun { |
| 426 | y: f32, |
| 427 | pen: f64, // where the viewer's pen stands after the last glyph, points |
| 428 | items: String, |
| 429 | hex: bool, // a hex string is open in `items` |
| 430 | adj: bool, // `items` holds a number, so it needs the array form |
| 431 | } |
| 432 | |
| 433 | impl TextState { |
| 434 | |
| 435 | fn flush(&mut self, s: &mut String) { |
| 436 | if let Some(mut run) = self.run.take() { |
| 437 | if run.hex { |
| 438 | run.items.push('>'); |
| 439 | } |
| 440 | if run.adj { |
| 441 | s.push_str(&fmt!("[{}] TJ\n", run.items)); |
| 442 | } else { |
| 443 | s.push_str(&fmt!("{} Tj\n", run.items)); |
| 444 | } |
| 445 | } |
| 446 | } |
| 447 | |
| 448 | fn close(&mut self, s: &mut String) { |
| 449 | self.flush(s); |
| 450 | if self.open { |
| 451 | s.push_str("ET\n"); |
| 452 | } |
| 453 | self.open = false; |
| 454 | self.font = None; |
| 455 | self.colour = None; |
| 456 | } |
| 457 | |
| 458 | /// Opens a text object if none is, and sets the colour and font when they differ from what is set. |
| 459 | fn begin( |
| 460 | &mut self, |
| 461 | s: &mut String, |
| 462 | translucent: bool, |
| 463 | alpha: &mut Option<u8>, |
| 464 | colour: Rgba, |
| 465 | font: &str, |
| 466 | size: f32, |
| 467 | ) { |
| 468 | if !self.open { |
| 469 | s.push_str("BT\n"); |
| 470 | self.open = true; |
| 471 | } |
| 472 | if self.colour != Some(colour) { |
| 473 | if translucent { |
| 474 | set_alpha(s, alpha, colour.a); |
| 475 | } |
| 476 | s.push_str(&fmt!("{} {} {} rg\n", chan(colour.r), chan(colour.g), chan(colour.b))); |
| 477 | self.colour = Some(colour); |
| 478 | } |
| 479 | let want = (font.to_string(), size); |
| 480 | if self.font.as_ref() != Some(&want) { |
| 481 | s.push_str(&fmt!("/{} {} Tf\n", font, numf32(size))); |
| 482 | self.font = Some(want); |
| 483 | } |
| 484 | } |
| 485 | } |
| 486 | |
| 487 | /// One Type-3 font: up to 256 distinct glyphs, and the object number reserved for its font dictionary |
| 488 | /// when the font was first needed. The `CharProc`, `/ToUnicode` and dictionary objects are written in |
| 489 | /// [`PdfStream::finish`]. |
| 490 | struct Type3Font { |
| 491 | obj: usize, // reserved object number of the font dictionary |
| 492 | glyphs: Vec<GlyphEntry>, // one per code, indexed by code (0..len) |
| 493 | } |
| 494 | |
| 495 | /// One stored glyph: its `CharProc` path operators, its advance and bounding box in the glyph frame |
| 496 | /// (at [`GLYPH_UPM`] units per em), and the source scalar(s) it stands for. |
| 497 | struct GlyphEntry { |
| 498 | ops: String, // path-construction operators of the outline |
| 499 | wx: i64, // advance width, glyph units |
| 500 | bbox: (f32, f32, f32, f32), // llx, lly, urx, ury, glyph units |
| 501 | text: String, // source scalar(s), for /ToUnicode; empty when unknown |
| 502 | } |
| 503 | |
| 504 | impl<W: Write> PdfStream<W> { |
| 505 | /// Opens a stream for a document of exactly `n` pages, writing the header, catalogue and page tree |
| 506 | /// at once. `compress` zlib-compresses each content stream, matching [`PdfWriter::with_compression`]. |
| 507 | pub fn new(out: W, n: usize, compress: bool) -> Outcome<Self> { |
| 508 | Self::new_with_outline(out, n, compress, Vec::new()) |
| 509 | } |
| 510 | |
| 511 | /// As [`new`](Self::new), but the file also carries a document outline (the viewer's bookmark side |
| 512 | /// panel). Each entry's page must be one of the `n` promised, since its destination names that page |
| 513 | /// object. An empty outline yields a file byte-identical to [`new`]'s. |
| 514 | pub fn new_with_outline(out: W, n: usize, compress: bool, outline: Vec<OutlineItem>) -> Outcome<Self> { |
| 515 | for it in &outline { |
| 516 | if it.page >= n { |
| 517 | return Err(err!( |
| 518 | "An outline entry points at page {} (zero-based), but the document has only {} \ |
| 519 | page(s); its destination could not be named.", it.page, n; Input, Invalid, Range)); |
| 520 | } |
| 521 | } |
| 522 | let obj_count = 2 + 2 * n; |
| 523 | let has_outline = !outline.is_empty(); |
| 524 | // The outline dict and one object per entry sit directly after the fixed page/content block; any |
| 525 | // image XObject is numbered past them. With no outline the numbering is exactly the original. |
| 526 | let outline_root = if has_outline { obj_count + 1 } else { 0 }; |
| 527 | let next_extra = if has_outline { |
| 528 | outline_root + 1 + outline.len() |
| 529 | } else { |
| 530 | obj_count + 1 |
| 531 | }; |
| 532 | let mut s = Self { |
| 533 | out, |
| 534 | compress, |
| 535 | n, |
| 536 | offsets: vec![0; obj_count + 1], |
| 537 | pos: 0, |
| 538 | added: 0, |
| 539 | next_extra, |
| 540 | hash_a: FNV_BASIS_A, |
| 541 | hash_b: FNV_BASIS_B, |
| 542 | outline, |
| 543 | outline_root, |
| 544 | glyph_slots: HashMap::new(), |
| 545 | fonts: Vec::new(), |
| 546 | cid_slots: HashMap::new(), |
| 547 | cid_fonts: Vec::new(), |
| 548 | }; |
| 549 | |
| 550 | res!(s.body(b"%PDF-1.7\n")); |
| 551 | // A comment of high bytes tells a naive tool the file is binary, so it is not mangled in |
| 552 | // transit. Four bytes above 127, as the specification suggests. |
| 553 | res!(s.body(b"%\xE2\xE3\xCF\xD3\n")); |
| 554 | |
| 555 | // The catalogue. When the file carries an outline the catalogue names it and asks the viewer to |
| 556 | // open the bookmark panel; with none it is byte for byte the original. |
| 557 | s.offsets[1] = s.pos; |
| 558 | if has_outline { |
| 559 | let cat = fmt!( |
| 560 | "1 0 obj\n<< /Type /Catalog /Pages 2 0 R /Outlines {} 0 R /PageMode /UseOutlines >>\nendobj\n", |
| 561 | outline_root); |
| 562 | res!(s.body(cat.as_bytes())); |
| 563 | } else { |
| 564 | res!(s.body(b"1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")); |
| 565 | } |
| 566 | |
| 567 | // The page tree, naming every page object up front, which is why `n` is fixed here. |
| 568 | s.offsets[2] = s.pos; |
| 569 | let mut kids = String::new(); |
| 570 | for i in 0..n { |
| 571 | if i > 0 { |
| 572 | kids.push(' '); |
| 573 | } |
| 574 | kids.push_str(&fmt!("{} 0 R", 3 + 2 * i)); |
| 575 | } |
| 576 | let tree = fmt!("2 0 obj\n<< /Type /Pages /Kids [{}] /Count {} >>\nendobj\n", kids, n); |
| 577 | res!(s.body(tree.as_bytes())); |
| 578 | |
| 579 | Ok(s) |
| 580 | } |
| 581 | |
| 582 | /// Writes the next page -- its page object and its content stream -- then advances. The page must |
| 583 | /// be the next of the `n` promised at construction; an extra page is a mismatch the file could not |
| 584 | /// name, so it is refused rather than written past the page tree. |
| 585 | /// |
| 586 | /// The content stream is built here, on the writer's thread and in page order, so a glyph's Type-3 |
| 587 | /// code and its font's object number are assigned once and deterministically -- two runs of the same |
| 588 | /// document yield the same bytes. The glyph outlines the content references are stored in the writer's |
| 589 | /// font tables and written in [`finish`](Self::finish). |
| 590 | pub fn page(&mut self, page: &PdfPage) -> Outcome<()> { |
| 591 | if self.added >= self.n { |
| 592 | return Err(err!( |
| 593 | "A PdfStream opened for {} page(s) was handed a further page; the page tree cannot \ |
| 594 | name it.", self.n; Input, Invalid, Excessive)); |
| 595 | } |
| 596 | let i = self.added; |
| 597 | let page_obj = 3 + 2 * i; |
| 598 | let content_obj = 4 + 2 * i; |
| 599 | |
| 600 | // Assign object numbers for every image on the page -- one for the image itself, and one more for |
| 601 | // its soft mask when it carries translucency -- so the page's resource dictionary can name them |
| 602 | // before the streams are written. The numbers run past the fixed page/content block, growing the |
| 603 | // object count a no-image document never touches. |
| 604 | let mut img_objs: Vec<(usize, Option<usize>)> = Vec::new(); |
| 605 | for d in &page.draws { |
| 606 | if let Draw::Image { alpha, .. } = d { |
| 607 | let image_obj = self.next_extra; |
| 608 | self.next_extra += 1; |
| 609 | let smask_obj = if alpha.is_some() { |
| 610 | let m = self.next_extra; |
| 611 | self.next_extra += 1; |
| 612 | Some(m) |
| 613 | } else { |
| 614 | None |
| 615 | }; |
| 616 | img_objs.push((image_obj, smask_obj)); |
| 617 | } |
| 618 | } |
| 619 | |
| 620 | // The link annotations, if any, take one further object -- the `/Annots` array -- numbered after the |
| 621 | // images. A page with no link claims no number and writes no `/Annots`, so its bytes are unchanged. |
| 622 | let annots_obj = if page.annots.is_empty() { |
| 623 | None |
| 624 | } else { |
| 625 | let a = self.next_extra; |
| 626 | self.next_extra += 1; |
| 627 | Some(a) |
| 628 | }; |
| 629 | |
| 630 | // Build the content stream, which assigns each new glyph a Type-3 code and reserves a font-dictionary |
| 631 | // object number the first time a font is needed. The used fonts come back so the page's resources can |
| 632 | // name them by forward reference, exactly as the page tree forward-references its pages. |
| 633 | let (raw, page_fonts) = res!(self.build_content(page)); |
| 634 | |
| 635 | // The content stream is now serialised, so its `/Length` is known before the object that wraps it; |
| 636 | // only the optional compression is left to do here. |
| 637 | let bytes = if self.compress { |
| 638 | res!(deflate(&raw)) |
| 639 | } else { |
| 640 | raw |
| 641 | }; |
| 642 | |
| 643 | self.offsets[page_obj] = self.pos; |
| 644 | let annots = match annots_obj { |
| 645 | Some(a) => fmt!(" /Annots {} 0 R", a), |
| 646 | None => String::new(), |
| 647 | }; |
| 648 | let head = fmt!( |
| 649 | "{} 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 {} {}] {} \ |
| 650 | /Contents {} 0 R{} >>\nendobj\n", |
| 651 | page_obj, numf(page.width), numf(page.height), |
| 652 | resources(page, &img_objs, &page_fonts), content_obj, annots); |
| 653 | res!(self.body(head.as_bytes())); |
| 654 | |
| 655 | self.offsets[content_obj] = self.pos; |
| 656 | let filter = if self.compress { " /Filter /FlateDecode" } else { "" }; |
| 657 | let open = fmt!( |
| 658 | "{} 0 obj\n<< /Length {}{} >>\nstream\n", content_obj, bytes.len(), filter); |
| 659 | res!(self.body(open.as_bytes())); |
| 660 | res!(self.body(&bytes)); |
| 661 | res!(self.body(b"\nendstream\nendobj\n")); |
| 662 | |
| 663 | // The image XObjects, in the order their numbers were assigned. The soft mask, when present, is |
| 664 | // written straight after the image object that references it. |
| 665 | let mut idx = 0; |
| 666 | for d in &page.draws { |
| 667 | if let Draw::Image { rgb, alpha, iw, ih, .. } = d { |
| 668 | let (image_obj, smask_obj) = img_objs[idx]; |
| 669 | idx += 1; |
| 670 | res!(self.write_image(image_obj, rgb, *iw, *ih, smask_obj)); |
| 671 | if let (Some(m), Some(a)) = (smask_obj, alpha) { |
| 672 | res!(self.write_smask(m, a, *iw, *ih)); |
| 673 | } |
| 674 | } |
| 675 | } |
| 676 | |
| 677 | // The `/Annots` array, on the number reserved above: one `/Link` annotation per link, each with its |
| 678 | // rectangle flipped into PDF's y-up frame and a URI action. Written after the images so the extra |
| 679 | // offsets are recorded in increasing number order. |
| 680 | if let Some(a) = annots_obj { |
| 681 | res!(self.write_annots(a, page)); |
| 682 | } |
| 683 | |
| 684 | self.added += 1; |
| 685 | Ok(()) |
| 686 | } |
| 687 | |
| 688 | /// Builds one page's content stream: the flip, then every shape's paint. A vector fill or stroke is |
| 689 | /// written inline; an embedded glyph joins a `TJ` run; an outline glyph is shown by a Type-3 text |
| 690 | /// operator against a code assigned here. Returns the serialised bytes and the fonts the page used, by |
| 691 | /// resource name and object number, so the caller can name them in the page's `/Resources`. |
| 692 | /// |
| 693 | /// Codes and fonts are assigned in draw order, on this (the writer's) thread and in page order across |
| 694 | /// the document, so the assignment is a deterministic function of the page sequence. Consecutive |
| 695 | /// glyphs share a `BT ... ET` text object while nothing else is drawn between them. |
| 696 | fn build_content(&mut self, page: &PdfPage) -> Outcome<(Vec<u8>, Vec<(String, usize)>)> { |
| 697 | let mut s = String::new(); |
| 698 | s.push_str(&fmt!("1 0 0 -1 0 {} cm\n", numf(page.height))); |
| 699 | |
| 700 | let translucent = page.draws.iter().any(|d| d.colour().a != 255); |
| 701 | let mut cur_alpha: Option<u8> = None; |
| 702 | let mut img_k = 0; // the image index, naming each `/Im{k}` XObject in draw order |
| 703 | let mut used: Vec<(String, usize)> = Vec::new(); |
| 704 | let mut ts = TextState::default(); |
| 705 | |
| 706 | for d in &page.draws { |
| 707 | // A non-text draw ends any text object first, so it is well formed. |
| 708 | if !matches!(d, Draw::Glyph { .. } | Draw::Text { .. }) { |
| 709 | ts.close(&mut s); |
| 710 | } |
| 711 | match d { |
| 712 | Draw::Image { x, y, w, h, .. } => { |
| 713 | // The page CTM already flips y into the engine's top-left frame. An image's sample space |
| 714 | // paints the unit square with its top row at the square's top, so mapping it into the |
| 715 | // rectangle at top-left (x, y) needs `w 0 0 -h x (y+h)`: the negative height and the raised |
| 716 | // origin put the first row at y and the last at y+h. Bracketed in q/Q so it disturbs nothing |
| 717 | // after it. |
| 718 | s.push_str("q\n"); |
| 719 | s.push_str(&fmt!("{} 0 0 {} {} {} cm\n", numf(*w), numf(-*h), numf(*x), numf(*y + *h))); |
| 720 | s.push_str(&fmt!("/Im{} Do\n", img_k)); |
| 721 | s.push_str("Q\n"); |
| 722 | img_k += 1; |
| 723 | }, |
| 724 | Draw::Fill { path, colour } => { |
| 725 | if translucent { |
| 726 | set_alpha(&mut s, &mut cur_alpha, colour.a); |
| 727 | } |
| 728 | s.push_str(&fmt!("{} {} {} rg\n", |
| 729 | chan(colour.r), chan(colour.g), chan(colour.b))); |
| 730 | path_ops(&mut s, path); |
| 731 | // Non-zero winding, to match the SVG writer, whose fill-rule defaults to nonzero. |
| 732 | s.push_str("f\n"); |
| 733 | }, |
| 734 | Draw::Stroke { path, colour, width } => { |
| 735 | if translucent { |
| 736 | set_alpha(&mut s, &mut cur_alpha, colour.a); |
| 737 | } |
| 738 | s.push_str(&fmt!("{} {} {} RG\n", |
| 739 | chan(colour.r), chan(colour.g), chan(colour.b))); |
| 740 | s.push_str(&fmt!("{} w\n", numf(*width))); |
| 741 | path_ops(&mut s, path); |
| 742 | s.push_str("S\n"); |
| 743 | }, |
| 744 | Draw::Glyph { outline, x, y, size, adv, colour, text } => { |
| 745 | let (font_idx, code) = self.assign_glyph(outline, *adv, *size, text); |
| 746 | let name = fmt!("F{}", font_idx); |
| 747 | let obj = self.fonts[font_idx].obj; |
| 748 | if !used.iter().any(|(n, _)| n == &name) { |
| 749 | used.push((name.clone(), obj)); |
| 750 | } |
| 751 | ts.flush(&mut s); |
| 752 | // Under `d1` a Type-3 glyph paints with the text state's fill colour, set by `begin`. |
| 753 | ts.begin(&mut s, translucent, &mut cur_alpha, *colour, &name, *size); |
| 754 | // The text matrix places the glyph and flips it back to y up: the page CTM flips the whole |
| 755 | // page in y, and this `[1 0 0 -1 x y]` flips the text within it, so the glyph reads upright. |
| 756 | // A per-glyph matrix means the font's advance never moves the pen -- the offset is exact. |
| 757 | s.push_str(&fmt!("1 0 0 -1 {} {} Tm\n", numf32(*x), numf32(*y))); |
| 758 | s.push_str(&fmt!("<{:02x}> Tj\n", code)); |
| 759 | }, |
| 760 | Draw::Text { font, gid, x, y, size, colour, text } => { |
| 761 | let idx = self.assign_text(font, *gid, text); |
| 762 | let name = fmt!("C{}", idx); |
| 763 | let obj = self.cid_fonts[idx].obj; |
| 764 | if !used.iter().any(|(n, _)| n == &name) { |
| 765 | used.push((name.clone(), obj)); |
| 766 | } |
| 767 | let same_state = ts.open |
| 768 | && ts.colour == Some(*colour) |
| 769 | && ts.font.as_ref().map_or(false, |(n, z)| n == &name && z == size); |
| 770 | let same_line = same_state && ts.run.as_ref().map_or(false, |r| r.y == *y); |
| 771 | if !same_line { |
| 772 | ts.flush(&mut s); |
| 773 | ts.begin(&mut s, translucent, &mut cur_alpha, *colour, &name, *size); |
| 774 | // One text matrix per run, flipped back to y up within the page's flip; the glyphs |
| 775 | // after the first are placed by the font's advances and the run's adjustments. |
| 776 | s.push_str(&fmt!("1 0 0 -1 {} {} Tm\n", numf32(*x), numf32(*y))); |
| 777 | ts.run = Some(TjRun { |
| 778 | y: *y, |
| 779 | pen: ((*x as f64) * 1000.0).round() / 1000.0, // as the `Tm` above writes it |
| 780 | items: String::new(), |
| 781 | hex: false, |
| 782 | adj: false, |
| 783 | }); |
| 784 | } |
| 785 | if let Some(run) = ts.run.as_mut() { |
| 786 | let z = *size as f64; |
| 787 | // The number that moves the pen from where the last advance left it to this glyph's |
| 788 | // placed x, in thousandths of the em, positive to the left as `TJ` has it. |
| 789 | let adj = if z > 0.0 { ((run.pen - *x as f64) * 1000.0 / z).round() as i64 } else { 0 }; |
| 790 | if adj != 0 { |
| 791 | if run.hex { |
| 792 | run.items.push('>'); |
| 793 | run.hex = false; |
| 794 | } |
| 795 | run.items.push_str(&fmt!("{}", adj)); |
| 796 | run.adj = true; |
| 797 | run.pen -= adj as f64 * z / 1000.0; |
| 798 | } |
| 799 | if !run.hex { |
| 800 | run.items.push('<'); |
| 801 | run.hex = true; |
| 802 | } |
| 803 | run.items.push_str(&fmt!("{:04X}", gid)); |
| 804 | run.pen += font.width(*gid) as f64 * z / 1000.0; |
| 805 | } |
| 806 | }, |
| 807 | } |
| 808 | } |
| 809 | ts.close(&mut s); |
| 810 | Ok((s.into_bytes(), used)) |
| 811 | } |
| 812 | |
| 813 | /// Records glyph `gid` of an embedded font as shown, reserving the font's `Type0` dictionary number |
| 814 | /// the first time the program is met. The first non-empty source text a glyph arrives with is kept. |
| 815 | fn assign_text(&mut self, font: &Arc<FontProgram>, gid: u16, text: &str) -> usize { |
| 816 | let idx = match self.cid_slots.get(&font.key()) { |
| 817 | Some(&i) => i, |
| 818 | None => { |
| 819 | let obj = self.next_extra; |
| 820 | self.next_extra += 1; |
| 821 | self.cid_fonts.push(CidFont { obj, prog: font.clone(), used: BTreeMap::new() }); |
| 822 | let i = self.cid_fonts.len() - 1; |
| 823 | self.cid_slots.insert(font.key(), i); |
| 824 | i |
| 825 | }, |
| 826 | }; |
| 827 | let entry = self.cid_fonts[idx].used.entry(gid).or_default(); |
| 828 | if entry.is_empty() && !text.is_empty() { |
| 829 | *entry = text.to_string(); |
| 830 | } |
| 831 | idx |
| 832 | } |
| 833 | |
| 834 | /// Assigns a glyph its Type-3 font index and code, storing the outline once. The key is the content of |
| 835 | /// the outline's path operators, so the same shape drawn by any face or size shares one `CharProc`. A |
| 836 | /// new glyph takes the next code; each font holds 256 codes, and crossing that boundary reserves a |
| 837 | /// fresh font-dictionary object number from the extra-object counter. |
| 838 | fn assign_glyph(&mut self, outline: &Path, adv: f32, size: f32, text: &str) -> (usize, u8) { |
| 839 | let mut ops = String::new(); |
| 840 | path_ops(&mut ops, outline); |
| 841 | let key = fnv1a(ops.as_bytes()); |
| 842 | if let Some(&slot) = self.glyph_slots.get(&key) { |
| 843 | return slot; |
| 844 | } |
| 845 | let seq = self.glyph_slots.len(); |
| 846 | let font_idx = seq / 256; |
| 847 | let code = (seq % 256) as u8; |
| 848 | if code == 0 { |
| 849 | // A new font begins: reserve its dictionary object now so a page written before `finish` can name |
| 850 | // it, and write the dictionary itself later. |
| 851 | let obj = self.next_extra; |
| 852 | self.next_extra += 1; |
| 853 | self.fonts.push(Type3Font { obj, glyphs: Vec::new() }); |
| 854 | } |
| 855 | let bbox = outline.bounds(&Transform::IDENTITY) |
| 856 | .map(|b| (b.x0, b.y0, b.x1, b.y1)) |
| 857 | .unwrap_or((0.0, 0.0, 0.0, 0.0)); |
| 858 | // The advance in glyph units: points at the shown size scaled to the em. Positioning never uses it, |
| 859 | // but the /Widths array and the `d1` operator declare it. |
| 860 | let wx = if size != 0.0 { (adv / size * GLYPH_UPM).round() as i64 } else { 0 }; |
| 861 | self.fonts[font_idx].glyphs.push(GlyphEntry { ops, wx, bbox, text: text.to_string() }); |
| 862 | self.glyph_slots.insert(key, (font_idx, code)); |
| 863 | (font_idx, code) |
| 864 | } |
| 865 | |
| 866 | /// Writes the Type-3 fonts: for each, its `CharProc` glyph streams, a `/ToUnicode` CMap, and the font |
| 867 | /// dictionary naming both. Called from [`finish`](Self::finish) after the last page, on the object |
| 868 | /// numbers reserved as each font and glyph was first met, so the cross-reference table stays exact. |
| 869 | fn write_fonts(&mut self) -> Outcome<()> { |
| 870 | let fonts = std::mem::take(&mut self.fonts); |
| 871 | for font in &fonts { |
| 872 | // Each glyph's outline is one CharProc stream. The `d1` operator declares the glyph a shape only, |
| 873 | // so it paints with the text state's fill colour; the advance and bounding box are its operands. |
| 874 | let mut cp_objs = Vec::with_capacity(font.glyphs.len()); |
| 875 | for g in &font.glyphs { |
| 876 | let obj = self.next_extra; |
| 877 | self.next_extra += 1; |
| 878 | cp_objs.push(obj); |
| 879 | let body = fmt!("{} 0 {} {} {} {} d1\n{}f\n", |
| 880 | g.wx, numf32(g.bbox.0), numf32(g.bbox.1), numf32(g.bbox.2), numf32(g.bbox.3), g.ops); |
| 881 | res!(self.write_stream(obj, body.as_bytes())); |
| 882 | } |
| 883 | |
| 884 | // The /ToUnicode CMap, so a viewer extracts the source text rather than the Type-3 codes. |
| 885 | let tu_obj = self.next_extra; |
| 886 | self.next_extra += 1; |
| 887 | let cmap = build_tounicode(&font.glyphs); |
| 888 | res!(self.write_stream(tu_obj, cmap.as_bytes())); |
| 889 | |
| 890 | // The font dictionary, on the number reserved when the font began. |
| 891 | self.set_extra_offset(font.obj); |
| 892 | let last = font.glyphs.len().saturating_sub(1); |
| 893 | // The font bounding box is the union of the glyph boxes. Start from the extremes so a glyph whose |
| 894 | // box is wholly positive or wholly negative is not clipped to the origin. |
| 895 | let mut bbox = (f32::MAX, f32::MAX, f32::MIN, f32::MIN); |
| 896 | for g in &font.glyphs { |
| 897 | bbox.0 = bbox.0.min(g.bbox.0); |
| 898 | bbox.1 = bbox.1.min(g.bbox.1); |
| 899 | bbox.2 = bbox.2.max(g.bbox.2); |
| 900 | bbox.3 = bbox.3.max(g.bbox.3); |
| 901 | } |
| 902 | let mut char_procs = String::new(); |
| 903 | let mut diffs = String::from("0"); |
| 904 | let mut widths = String::new(); |
| 905 | for (code, obj) in cp_objs.iter().enumerate() { |
| 906 | char_procs.push_str(&fmt!(" /g{} {} 0 R", code, obj)); |
| 907 | diffs.push_str(&fmt!(" /g{}", code)); |
| 908 | if code > 0 { |
| 909 | widths.push(' '); |
| 910 | } |
| 911 | widths.push_str(&fmt!("{}", font.glyphs[code].wx)); |
| 912 | } |
| 913 | let dict = fmt!( |
| 914 | "{} 0 obj\n<< /Type /Font /Subtype /Type3 \ |
| 915 | /FontBBox [{} {} {} {}] /FontMatrix [0.001 0 0 0.001 0 0] \ |
| 916 | /CharProcs <<{} >> /Encoding << /Type /Encoding /Differences [{}] >> \ |
| 917 | /FirstChar 0 /LastChar {} /Widths [{}] /ToUnicode {} 0 R /Resources << >> >>\nendobj\n", |
| 918 | font.obj, |
| 919 | numf32(bbox.0), numf32(bbox.1), numf32(bbox.2), numf32(bbox.3), |
| 920 | char_procs, diffs, last, widths, tu_obj); |
| 921 | res!(self.body(dict.as_bytes())); |
| 922 | } |
| 923 | Ok(()) |
| 924 | } |
| 925 | |
| 926 | /// Writes the embedded fonts: for each program, a `Type0` font over one `CIDFont` whose CIDs are the |
| 927 | /// program's glyph ids (`Identity-H` encoding, and for TrueType an identity `/CIDToGIDMap`), a font |
| 928 | /// descriptor naming the subset program, and a `/ToUnicode` CMap from each shown glyph to its text. |
| 929 | fn write_cid_fonts(&mut self) -> Outcome<()> { |
| 930 | let fonts = std::mem::take(&mut self.cid_fonts); |
| 931 | for font in &fonts { |
| 932 | let prog = &font.prog; |
| 933 | let gids: BTreeSet<u16> = font.used.keys().copied().collect(); |
| 934 | let file = prog.subset(&gids); |
| 935 | |
| 936 | // The six-letter subset tag, from the program and the glyph set, so it is stable across runs yet |
| 937 | // differs between two subsets of one font. |
| 938 | let mut h = FNV_BASIS_A ^ prog.key(); |
| 939 | for g in &gids { |
| 940 | for b in g.to_be_bytes() { |
| 941 | h ^= b as u64; |
| 942 | h = h.wrapping_mul(FNV_PRIME); |
| 943 | } |
| 944 | } |
| 945 | let mut tag = String::new(); |
| 946 | for _ in 0..6 { |
| 947 | tag.push((b'A' + (h % 26) as u8) as char); |
| 948 | h /= 26; |
| 949 | } |
| 950 | let base = fmt!("{}+{}", tag, prog.name()); |
| 951 | |
| 952 | let cid_obj = self.next_extra; |
| 953 | let fd_obj = cid_obj + 1; |
| 954 | let ff_obj = cid_obj + 2; |
| 955 | let tu_obj = cid_obj + 3; |
| 956 | self.next_extra += 4; |
| 957 | |
| 958 | self.set_extra_offset(font.obj); |
| 959 | let type0 = fmt!( |
| 960 | "{} 0 obj\n<< /Type /Font /Subtype /Type0 /BaseFont /{} /Encoding /Identity-H \ |
| 961 | /DescendantFonts [{} 0 R] /ToUnicode {} 0 R >>\nendobj\n", |
| 962 | font.obj, base, cid_obj, tu_obj); |
| 963 | res!(self.body(type0.as_bytes())); |
| 964 | |
| 965 | // Widths, one bracketed run per stretch of consecutive glyph ids. |
| 966 | let mut w = String::new(); |
| 967 | let mut prev: Option<u16> = None; |
| 968 | for &g in &gids { |
| 969 | match prev { |
| 970 | Some(p) if p.checked_add(1) == Some(g) => w.push_str(&fmt!(" {}", prog.width(g))), |
| 971 | Some(_) => w.push_str(&fmt!("] {} [{}", g, prog.width(g))), |
| 972 | None => w.push_str(&fmt!("{} [{}", g, prog.width(g))), |
| 973 | } |
| 974 | prev = Some(g); |
| 975 | } |
| 976 | if prev.is_some() { |
| 977 | w.push(']'); |
| 978 | } |
| 979 | let (subtype, extra) = match prog.outlines() { |
| 980 | Outlines::TrueType => ("CIDFontType2", " /CIDToGIDMap /Identity"), |
| 981 | Outlines::Cff => ("CIDFontType0", ""), |
| 982 | }; |
| 983 | let cid = fmt!( |
| 984 | "{} 0 obj\n<< /Type /Font /Subtype /{} /BaseFont /{} \ |
| 985 | /CIDSystemInfo << /Registry (Adobe) /Ordering (Identity) /Supplement 0 >> \ |
| 986 | /FontDescriptor {} 0 R /W [{}]{} >>\nendobj\n", |
| 987 | cid_obj, subtype, base, fd_obj, w, extra); |
| 988 | self.set_extra_offset(cid_obj); |
| 989 | res!(self.body(cid.as_bytes())); |
| 990 | |
| 991 | // Symbolic (bit 3), since the glyphs are reached by id rather than a standard encoding; fixed |
| 992 | // pitch (bit 1) and italic (bit 7) as the program declares them. |
| 993 | let mut flags = 4u32; |
| 994 | if prog.fixed_pitch() { flags |= 1; } |
| 995 | if prog.italic_angle() != 0.0 { flags |= 64; } |
| 996 | let bb = prog.bbox(); |
| 997 | let stem_v = if prog.weight() >= 600 { 120 } else { 80 }; |
| 998 | let file_key = match file { |
| 999 | FontFile::TrueType(_) => "FontFile2", |
| 1000 | _ => "FontFile3", |
| 1001 | }; |
| 1002 | let fd = fmt!( |
| 1003 | "{} 0 obj\n<< /Type /FontDescriptor /FontName /{} /Flags {} /FontBBox [{} {} {} {}] \ |
| 1004 | /ItalicAngle {} /Ascent {} /Descent {} /CapHeight {} /StemV {} /{} {} 0 R >>\nendobj\n", |
| 1005 | fd_obj, base, flags, bb[0], bb[1], bb[2], bb[3], numf32(prog.italic_angle()), |
| 1006 | prog.ascent(), prog.descent(), prog.cap_height(), stem_v, file_key, ff_obj); |
| 1007 | self.set_extra_offset(fd_obj); |
| 1008 | res!(self.body(fd.as_bytes())); |
| 1009 | |
| 1010 | // The program itself, always compressed: it is binary whatever the content streams are. |
| 1011 | let (data, dict) = match &file { |
| 1012 | FontFile::TrueType(b) => (b, fmt!("/Length1 {}", b.len())), |
| 1013 | FontFile::Cff(b) => (b, "/Subtype /CIDFontType0C".to_string()), |
| 1014 | FontFile::OpenType(b) => (b, "/Subtype /OpenType".to_string()), |
| 1015 | }; |
| 1016 | let packed = res!(deflate(data)); |
| 1017 | self.set_extra_offset(ff_obj); |
| 1018 | let head = fmt!("{} 0 obj\n<< {} /Length {} /Filter /FlateDecode >>\nstream\n", |
| 1019 | ff_obj, dict, packed.len()); |
| 1020 | res!(self.body(head.as_bytes())); |
| 1021 | res!(self.body(&packed)); |
| 1022 | res!(self.body(b"\nendstream\nendobj\n")); |
| 1023 | |
| 1024 | let maps: Vec<(u32, &str)> = font.used.iter() |
| 1025 | .filter(|(_, t)| !t.is_empty()) |
| 1026 | .map(|(g, t)| (*g as u32, t.as_str())) |
| 1027 | .collect(); |
| 1028 | let cmap = tounicode_cmap(2, &maps); |
| 1029 | res!(self.write_stream(tu_obj, cmap.as_bytes())); |
| 1030 | } |
| 1031 | Ok(()) |
| 1032 | } |
| 1033 | |
| 1034 | /// Writes one body stream object, compressing it and marking `/FlateDecode` when compression is on, and |
| 1035 | /// records its offset. The bytes are folded into the deterministic `/ID` like all body bytes. |
| 1036 | fn write_stream(&mut self, obj: usize, raw: &[u8]) -> Outcome<()> { |
| 1037 | let bytes = if self.compress { |
| 1038 | res!(deflate(raw)) |
| 1039 | } else { |
| 1040 | raw.to_vec() |
| 1041 | }; |
| 1042 | self.set_extra_offset(obj); |
| 1043 | let filter = if self.compress { " /Filter /FlateDecode" } else { "" }; |
| 1044 | let head = fmt!("{} 0 obj\n<< /Length {}{} >>\nstream\n", obj, bytes.len(), filter); |
| 1045 | res!(self.body(head.as_bytes())); |
| 1046 | res!(self.body(&bytes)); |
| 1047 | res!(self.body(b"\nendstream\nendobj\n")); |
| 1048 | Ok(()) |
| 1049 | } |
| 1050 | |
| 1051 | /// Writes a page's `/Annots` array: one `/Link` annotation per [`LinkAnnot`], its rectangle flipped from |
| 1052 | /// the engine's top-left, y-down frame into PDF's bottom-left, y-up one, its border drawn away, and its |
| 1053 | /// action opening the URI. The dictionaries are inlined in the array object, so a page's links cost one |
| 1054 | /// object however many they are. Only called when the page carries at least one link. |
| 1055 | fn write_annots(&mut self, obj: usize, page: &PdfPage) -> Outcome<()> { |
| 1056 | self.set_extra_offset(obj); |
| 1057 | let mut s = fmt!("{} 0 obj\n[ ", obj); |
| 1058 | for a in &page.annots { |
| 1059 | // Flip y: a point at engine y lands at PDF `height - y`, so the top edge (smaller engine y) |
| 1060 | // becomes the upper PDF coordinate and the bottom edge the lower one. |
| 1061 | let lly = page.height - a.y1; |
| 1062 | let ury = page.height - a.y0; |
| 1063 | s.push_str(&fmt!( |
| 1064 | "<< /Type /Annot /Subtype /Link /Border [0 0 0] /Rect [{} {} {} {}] \ |
| 1065 | /A << /S /URI /URI {} >> >> ", |
| 1066 | numf(a.x0), numf(lly), numf(a.x1), numf(ury), pdf_text_string(&a.uri))); |
| 1067 | } |
| 1068 | s.push_str("]\nendobj\n"); |
| 1069 | res!(self.body(s.as_bytes())); |
| 1070 | Ok(()) |
| 1071 | } |
| 1072 | |
| 1073 | /// Writes an image XObject: a straight-RGB, eight-bit `/DeviceRGB` sample stream, always |
| 1074 | /// zlib-compressed so a photograph does not bloat the file, and pointing at its soft mask when one |
| 1075 | /// was assigned. The samples are folded into the deterministic `/ID` like all body bytes. |
| 1076 | fn write_image( |
| 1077 | &mut self, |
| 1078 | obj: usize, |
| 1079 | rgb: &[u8], |
| 1080 | iw: usize, |
| 1081 | ih: usize, |
| 1082 | smask: Option<usize>, |
| 1083 | ) |
| 1084 | -> Outcome<()> |
| 1085 | { |
| 1086 | let data = res!(deflate(rgb)); |
| 1087 | self.set_extra_offset(obj); |
| 1088 | let mask = match smask { |
| 1089 | Some(m) => fmt!(" /SMask {} 0 R", m), |
| 1090 | None => String::new(), |
| 1091 | }; |
| 1092 | let head = fmt!( |
| 1093 | "{} 0 obj\n<< /Type /XObject /Subtype /Image /Width {} /Height {} /ColorSpace /DeviceRGB \ |
| 1094 | /BitsPerComponent 8{} /Filter /FlateDecode /Length {} >>\nstream\n", |
| 1095 | obj, iw, ih, mask, data.len()); |
| 1096 | res!(self.body(head.as_bytes())); |
| 1097 | res!(self.body(&data)); |
| 1098 | res!(self.body(b"\nendstream\nendobj\n")); |
| 1099 | Ok(()) |
| 1100 | } |
| 1101 | |
| 1102 | /// Writes a soft-mask XObject: a single-channel `/DeviceGray` image the same size as its owner, its |
| 1103 | /// samples the straight alpha, zlib-compressed and folded into the `/ID` like any body bytes. |
| 1104 | fn write_smask(&mut self, obj: usize, alpha: &[u8], iw: usize, ih: usize) -> Outcome<()> { |
| 1105 | let data = res!(deflate(alpha)); |
| 1106 | self.set_extra_offset(obj); |
| 1107 | let head = fmt!( |
| 1108 | "{} 0 obj\n<< /Type /XObject /Subtype /Image /Width {} /Height {} /ColorSpace /DeviceGray \ |
| 1109 | /BitsPerComponent 8 /Filter /FlateDecode /Length {} >>\nstream\n", |
| 1110 | obj, iw, ih, data.len()); |
| 1111 | res!(self.body(head.as_bytes())); |
| 1112 | res!(self.body(&data)); |
| 1113 | res!(self.body(b"\nendstream\nendobj\n")); |
| 1114 | Ok(()) |
| 1115 | } |
| 1116 | |
| 1117 | /// Writes the document outline: the `/Outlines` dictionary, then one object per entry, each a title, |
| 1118 | /// its tree links, and a destination fitting the top of the page it names. The tree is built from the |
| 1119 | /// flat, reading-order entry list by nesting each entry under the nearest preceding shallower one; the |
| 1120 | /// counts are shown open, so a viewer opens the whole tree. The objects take the numbers reserved at |
| 1121 | /// construction, directly after the fixed page/content block. |
| 1122 | fn write_outline(&mut self) -> Outcome<()> { |
| 1123 | let root = self.outline_root; |
| 1124 | let item_base = root + 1; // the first entry's object number |
| 1125 | // Taken out so the entry loop may borrow it while `self` is mutated for each object written. |
| 1126 | let outline = std::mem::take(&mut self.outline); |
| 1127 | let (links, roots) = build_outline_links(&outline); |
| 1128 | |
| 1129 | // The `/Outlines` dict: its first and last top-level entries, and the total number of entries, |
| 1130 | // all open. |
| 1131 | self.set_extra_offset(root); |
| 1132 | let first_root = roots.first().map(|&i| item_base + i).unwrap_or(0); |
| 1133 | let last_root = roots.last().map(|&i| item_base + i).unwrap_or(0); |
| 1134 | let head = fmt!( |
| 1135 | "{} 0 obj\n<< /Type /Outlines /First {} 0 R /Last {} 0 R /Count {} >>\nendobj\n", |
| 1136 | root, first_root, last_root, outline.len()); |
| 1137 | res!(self.body(head.as_bytes())); |
| 1138 | |
| 1139 | // One object per entry, in reading order so its reserved number matches its slice index. |
| 1140 | for (i, item) in outline.iter().enumerate() { |
| 1141 | let obj = item_base + i; |
| 1142 | let link = &links[i]; |
| 1143 | let parent = match link.parent { |
| 1144 | Some(p) => item_base + p, |
| 1145 | None => root, |
| 1146 | }; |
| 1147 | let page_obj = 3 + 2 * item.page; |
| 1148 | |
| 1149 | let mut dict = fmt!("{} 0 obj\n<< /Title {} /Parent {} 0 R", |
| 1150 | obj, pdf_text_string(&item.title), parent); |
| 1151 | if let Some(p) = link.prev { |
| 1152 | dict.push_str(&fmt!(" /Prev {} 0 R", item_base + p)); |
| 1153 | } |
| 1154 | if let Some(nx) = link.next { |
| 1155 | dict.push_str(&fmt!(" /Next {} 0 R", item_base + nx)); |
| 1156 | } |
| 1157 | if let (Some(f), Some(l)) = (link.first, link.last) { |
| 1158 | dict.push_str(&fmt!(" /First {} 0 R /Last {} 0 R /Count {}", |
| 1159 | item_base + f, item_base + l, link.count)); |
| 1160 | } |
| 1161 | dict.push_str(&fmt!(" /Dest [{} 0 R /Fit] >>\nendobj\n", page_obj)); |
| 1162 | self.set_extra_offset(obj); |
| 1163 | res!(self.body(dict.as_bytes())); |
| 1164 | } |
| 1165 | Ok(()) |
| 1166 | } |
| 1167 | |
| 1168 | /// Records the byte offset of an extra object -- an image or soft mask numbered past the fixed |
| 1169 | /// page/content block -- growing the offset table to reach it. The extras are assigned and written in |
| 1170 | /// increasing number order, so the table grows one slot at a time and stays indexed by object number. |
| 1171 | fn set_extra_offset(&mut self, obj: usize) { |
| 1172 | while self.offsets.len() <= obj { |
| 1173 | self.offsets.push(0); |
| 1174 | } |
| 1175 | self.offsets[obj] = self.pos; |
| 1176 | } |
| 1177 | |
| 1178 | /// Closes the file: writes the cross-reference table and the trailer, flushes, and returns the sink. |
| 1179 | /// The `/ID` is the body hash folded as the body was written, so the file matches |
| 1180 | /// [`PdfWriter::to_bytes`] to the byte. |
| 1181 | pub fn finish(mut self) -> Outcome<W> { |
| 1182 | // The outline objects -- the `/Outlines` dict and one object per entry -- are written after the |
| 1183 | // pages, on the numbers reserved for them at construction. A document with no outline writes none. |
| 1184 | if !self.outline.is_empty() { |
| 1185 | res!(self.write_outline()); |
| 1186 | } |
| 1187 | |
| 1188 | // The Type-3 fonts -- each glyph's CharProc, a /ToUnicode CMap, and the font dictionary -- are |
| 1189 | // written last, taking further object numbers past everything the pages reserved. A document with no |
| 1190 | // text writes none, so its numbering is exactly the original. |
| 1191 | if !self.fonts.is_empty() { |
| 1192 | res!(self.write_fonts()); |
| 1193 | } |
| 1194 | |
| 1195 | // The embedded fonts, each subset to the glyphs shown: its Type0 dictionary on the number reserved |
| 1196 | // at first use, then the CIDFont, descriptor, font file and /ToUnicode on fresh numbers. |
| 1197 | if !self.cid_fonts.is_empty() { |
| 1198 | res!(self.write_cid_fonts()); |
| 1199 | } |
| 1200 | |
| 1201 | // The fixed page/content block is `2 + 2n` objects; the outline, every image and soft mask, and the |
| 1202 | // Type-3 fonts took further numbers past it, so the highest object written is one below the next free |
| 1203 | // number. A document with no outline, image or text leaves `next_extra` at `2 + 2n + 1`, the original. |
| 1204 | let obj_count = self.next_extra - 1; |
| 1205 | |
| 1206 | // The identifier is derived from the body already written, never from the clock. The two halves |
| 1207 | // were folded byte by byte as the body streamed out. |
| 1208 | let id = fmt!("{:016x}{:016x}", self.hash_a, self.hash_b); |
| 1209 | |
| 1210 | // The cross-reference table and trailer sit after the body and are not part of the hash, so they |
| 1211 | // are written straight to the sink without folding. Every entry is exactly twenty bytes: a |
| 1212 | // ten-digit offset, a five-digit generation, the type, and a two-byte end. |
| 1213 | let xref_off = self.pos; |
| 1214 | let mut tail = String::new(); |
| 1215 | tail.push_str(&fmt!("xref\n0 {}\n", obj_count + 1)); |
| 1216 | tail.push_str("0000000000 65535 f\r\n"); |
| 1217 | for k in 1..=obj_count { |
| 1218 | tail.push_str(&fmt!("{:010} 00000 n\r\n", self.offsets[k])); |
| 1219 | } |
| 1220 | tail.push_str(&fmt!( |
| 1221 | "trailer\n<< /Size {} /Root 1 0 R /ID [<{}> <{}>] >>\nstartxref\n{}\n%%EOF\n", |
| 1222 | obj_count + 1, id, id, xref_off)); |
| 1223 | res!(self.out.write_all(tail.as_bytes())); |
| 1224 | res!(self.out.flush()); |
| 1225 | Ok(self.out) |
| 1226 | } |
| 1227 | |
| 1228 | /// Writes a run of body bytes: out to the sink, on to the running offset, and folded into both |
| 1229 | /// halves of the deterministic `/ID`. Only the body passes through here; the xref and trailer, which |
| 1230 | /// the hash excludes, are written directly. |
| 1231 | fn body(&mut self, bytes: &[u8]) -> Outcome<()> { |
| 1232 | res!(self.out.write_all(bytes)); |
| 1233 | self.pos += bytes.len(); |
| 1234 | for &b in bytes { |
| 1235 | self.hash_a ^= b as u64; |
| 1236 | self.hash_a = self.hash_a.wrapping_mul(FNV_PRIME); |
| 1237 | self.hash_b ^= b as u64; |
| 1238 | self.hash_b = self.hash_b.wrapping_mul(FNV_PRIME); |
| 1239 | } |
| 1240 | Ok(()) |
| 1241 | } |
| 1242 | } |
| 1243 | |
| 1244 | /// Builds the `/ToUnicode` CMap for a Type-3 font: one `bfchar` mapping per glyph whose source text is |
| 1245 | /// known, the code as a one-byte hex string. A glyph with no known source (a decoration, or the tail of a |
| 1246 | /// decomposed cluster) is left out. |
| 1247 | fn build_tounicode(glyphs: &[GlyphEntry]) -> String { |
| 1248 | let maps: Vec<(u32, &str)> = glyphs.iter().enumerate() |
| 1249 | .filter(|(_, g)| !g.text.is_empty()) |
| 1250 | .map(|(code, g)| (code as u32, g.text.as_str())) |
| 1251 | .collect(); |
| 1252 | tounicode_cmap(1, &maps) |
| 1253 | } |
| 1254 | |
| 1255 | /// A `/ToUnicode` CMap over codes `width` bytes wide, each mapped to its text as UTF-16BE, so a viewer |
| 1256 | /// extracts the real words rather than the font's private codes. A ligature maps its one code to every |
| 1257 | /// character it joins. The `bfchar` entries are batched under a hundred, the CMap operator's limit. |
| 1258 | fn tounicode_cmap(width: usize, maps: &[(u32, &str)]) -> String { |
| 1259 | let mut s = String::new(); |
| 1260 | s.push_str("/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n"); |
| 1261 | s.push_str("/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def\n"); |
| 1262 | s.push_str("/CMapName /Adobe-Identity-UCS def\n/CMapType 2 def\n"); |
| 1263 | if width == 1 { |
| 1264 | s.push_str("1 begincodespacerange\n<00> <FF>\nendcodespacerange\n"); |
| 1265 | } else { |
| 1266 | s.push_str("1 begincodespacerange\n<0000> <FFFF>\nendcodespacerange\n"); |
| 1267 | } |
| 1268 | for chunk in maps.chunks(100) { |
| 1269 | s.push_str(&fmt!("{} beginbfchar\n", chunk.len())); |
| 1270 | for (code, text) in chunk { |
| 1271 | let mut hex = String::new(); |
| 1272 | for u in text.encode_utf16() { |
| 1273 | hex.push_str(&fmt!("{:04X}", u)); |
| 1274 | } |
| 1275 | if width == 1 { |
| 1276 | s.push_str(&fmt!("<{:02X}> <{}>\n", code, hex)); |
| 1277 | } else { |
| 1278 | s.push_str(&fmt!("<{:04X}> <{}>\n", code, hex)); |
| 1279 | } |
| 1280 | } |
| 1281 | s.push_str("endbfchar\n"); |
| 1282 | } |
| 1283 | s.push_str("endcmap\nCMapName currentdict /CMap defineresource pop\nend\nend\n"); |
| 1284 | s |
| 1285 | } |
| 1286 | |
| 1287 | /// Emits the path-construction operators for one path. |
| 1288 | /// |
| 1289 | /// A move is `m`, a line `l`, a cubic `c`, a close `h`. A quadratic has no operator of its own and is |
| 1290 | /// elevated to a cubic exactly: the cubic through the same ends whose two controls sit two-thirds of |
| 1291 | /// the way from each end towards the quadratic's single control traces the identical curve. The |
| 1292 | /// current point is tracked because the elevation needs the segment's start, and a close returns it |
| 1293 | /// to where the contour began. |
| 1294 | fn path_ops(s: &mut String, path: &Path) { |
| 1295 | let mut cur = Pt::default(); |
| 1296 | let mut start = Pt::default(); |
| 1297 | for seg in path.segs() { |
| 1298 | match *seg { |
| 1299 | Seg::MoveTo(p) => { |
| 1300 | s.push_str(&fmt!("{} {} m\n", numf32(p.x), numf32(p.y))); |
| 1301 | cur = p; |
| 1302 | start = p; |
| 1303 | }, |
| 1304 | Seg::LineTo(p) => { |
| 1305 | s.push_str(&fmt!("{} {} l\n", numf32(p.x), numf32(p.y))); |
| 1306 | cur = p; |
| 1307 | }, |
| 1308 | Seg::QuadTo(c, p) => { |
| 1309 | let two_thirds = 2.0 / 3.0; |
| 1310 | let c0 = Pt::new( |
| 1311 | cur.x + two_thirds * (c.x - cur.x), |
| 1312 | cur.y + two_thirds * (c.y - cur.y)); |
| 1313 | let c1 = Pt::new( |
| 1314 | p.x + two_thirds * (c.x - p.x), |
| 1315 | p.y + two_thirds * (c.y - p.y)); |
| 1316 | s.push_str(&fmt!("{} {} {} {} {} {} c\n", |
| 1317 | numf32(c0.x), numf32(c0.y), numf32(c1.x), numf32(c1.y), |
| 1318 | numf32(p.x), numf32(p.y))); |
| 1319 | cur = p; |
| 1320 | }, |
| 1321 | Seg::CubicTo(c0, c1, p) => { |
| 1322 | s.push_str(&fmt!("{} {} {} {} {} {} c\n", |
| 1323 | numf32(c0.x), numf32(c0.y), numf32(c1.x), numf32(c1.y), |
| 1324 | numf32(p.x), numf32(p.y))); |
| 1325 | cur = p; |
| 1326 | }, |
| 1327 | Seg::Close => { |
| 1328 | s.push_str("h\n"); |
| 1329 | cur = start; |
| 1330 | }, |
| 1331 | } |
| 1332 | } |
| 1333 | } |
| 1334 | |
| 1335 | /// The page's `/Resources`: an `/ExtGState` for each distinct alpha when a shape is translucent, an |
| 1336 | /// `/XObject` dict naming each image `/Im{k}` by the object number assigned in [`PdfStream::page`], and a |
| 1337 | /// `/Font` dict naming each font this page shows text from, `/F{k}` for a Type-3 font and `/C{k}` for an |
| 1338 | /// embedded one. An all-opaque page with no |
| 1339 | /// image and no text carries an empty resource dictionary -- byte for byte the original. |
| 1340 | fn resources(page: &PdfPage, img_objs: &[(usize, Option<usize>)], page_fonts: &[(String, usize)]) -> String { |
| 1341 | let translucent = page.draws.iter().any(|d| d.colour().a != 255); |
| 1342 | |
| 1343 | // The image resource dict, `/Im{k}` in draw order to match the content stream's `Do` names. |
| 1344 | let xobjects = if img_objs.is_empty() { |
| 1345 | String::new() |
| 1346 | } else { |
| 1347 | let mut x = String::from(" /XObject << "); |
| 1348 | for (k, (obj, _)) in img_objs.iter().enumerate() { |
| 1349 | x.push_str(&fmt!("/Im{} {} 0 R ", k, obj)); |
| 1350 | } |
| 1351 | x.push_str(">>"); |
| 1352 | x |
| 1353 | }; |
| 1354 | |
| 1355 | // The font resource dict, by the names the content stream uses. |
| 1356 | let fonts = if page_fonts.is_empty() { |
| 1357 | String::new() |
| 1358 | } else { |
| 1359 | let mut f = String::from(" /Font << "); |
| 1360 | for (name, obj) in page_fonts { |
| 1361 | f.push_str(&fmt!("/{} {} 0 R ", name, obj)); |
| 1362 | } |
| 1363 | f.push_str(">>"); |
| 1364 | f |
| 1365 | }; |
| 1366 | |
| 1367 | if !translucent { |
| 1368 | if xobjects.is_empty() && fonts.is_empty() { |
| 1369 | return fmt!("/Resources << >>"); |
| 1370 | } |
| 1371 | return fmt!("/Resources <<{}{} >>", xobjects, fonts); |
| 1372 | } |
| 1373 | |
| 1374 | let mut alphas: Vec<u8> = Vec::new(); |
| 1375 | for d in &page.draws { |
| 1376 | let a = d.colour().a; |
| 1377 | if !alphas.contains(&a) { |
| 1378 | alphas.push(a); |
| 1379 | } |
| 1380 | } |
| 1381 | // The opaque state is always present, so a shape after a translucent one can return to full |
| 1382 | // opacity. |
| 1383 | if !alphas.contains(&255) { |
| 1384 | alphas.push(255); |
| 1385 | } |
| 1386 | alphas.sort_unstable(); |
| 1387 | let mut gs = String::new(); |
| 1388 | for a in &alphas { |
| 1389 | let v = chan(*a); |
| 1390 | gs.push_str(&fmt!("/GS{} << /ca {} /CA {} >> ", a, v, v)); |
| 1391 | } |
| 1392 | fmt!("/Resources << /ExtGState << {}>>{}{} >>", gs, xobjects, fonts) |
| 1393 | } |
| 1394 | |
| 1395 | /// Sets the alpha graphics state, but only when it changes, naming each state `/GSn` by its alpha |
| 1396 | /// byte to match [`resources`]. |
| 1397 | fn set_alpha(s: &mut String, cur: &mut Option<u8>, a: u8) { |
| 1398 | if *cur != Some(a) { |
| 1399 | s.push_str(&fmt!("/GS{} gs\n", a)); |
| 1400 | *cur = Some(a); |
| 1401 | } |
| 1402 | } |
| 1403 | |
| 1404 | /// One 8-bit channel as a PDF colour component from 0 to 1. |
| 1405 | fn chan(c: u8) -> String { |
| 1406 | dec6((c as f64) / 255.0) |
| 1407 | } |
| 1408 | |
| 1409 | /// A number to at most six decimal places, trailing zeros trimmed, for a colour or an alpha. |
| 1410 | fn dec6(v: f64) -> String { |
| 1411 | let s = fmt!("{:.6}", v); |
| 1412 | let t = s.trim_end_matches('0').trim_end_matches('.'); |
| 1413 | if t.is_empty() { "0".to_string() } else { t.to_string() } |
| 1414 | } |
| 1415 | |
| 1416 | /// A length or coordinate to three decimal places, trailing zeros and the point trimmed. Three places |
| 1417 | /// is a thousandth of a point -- far below any renderer's resolution -- so nothing visible is lost, |
| 1418 | /// while a coordinate like `841.8897705078125` shrinks to `841.89`: coordinate digits are about a |
| 1419 | /// third of a content stream's bytes, and full f32 precision spends them on noise. The rounding is |
| 1420 | /// deterministic (round-half-to-even), which the content-addressed file needs. |
| 1421 | fn numf(v: f64) -> String { |
| 1422 | trim_dec(fmt!("{:.3}", v)) |
| 1423 | } |
| 1424 | |
| 1425 | fn numf32(v: f32) -> String { |
| 1426 | trim_dec(fmt!("{:.3}", v)) |
| 1427 | } |
| 1428 | |
| 1429 | /// Trims the trailing zeros and any bare point from a fixed-precision decimal, and folds a rounded |
| 1430 | /// `-0` back to `0` so the bytes stay canonical. |
| 1431 | fn trim_dec(s: String) -> String { |
| 1432 | if !s.contains('.') { |
| 1433 | return s; |
| 1434 | } |
| 1435 | let t = s.trim_end_matches('0').trim_end_matches('.'); |
| 1436 | if t.is_empty() || t == "-0" { "0".to_string() } else { t.to_string() } |
| 1437 | } |
| 1438 | |
| 1439 | /// Zlib-compresses a content stream, for `/FlateDecode`. |
| 1440 | /// |
| 1441 | /// The level is fixed so the output is byte-deterministic for a given input and a given `flate2` |
| 1442 | /// version. |
| 1443 | fn deflate(raw: &[u8]) -> Outcome<Vec<u8>> { |
| 1444 | use flate2::write::ZlibEncoder; |
| 1445 | use flate2::Compression; |
| 1446 | use std::io::Write; |
| 1447 | let mut enc = ZlibEncoder::new(Vec::new(), Compression::new(6)); |
| 1448 | res!(enc.write_all(raw)); |
| 1449 | Ok(res!(enc.finish())) |
| 1450 | } |
| 1451 | |
| 1452 | #[cfg(test)] |
| 1453 | mod tests { |
| 1454 | use super::*; |
| 1455 | use crate::path::{ |
| 1456 | Bounds, |
| 1457 | PathBuilder, |
| 1458 | }; |
| 1459 | |
| 1460 | #[test] |
| 1461 | fn test_a_quadratic_elevates_to_the_matching_cubic_00() -> Outcome<()> { |
| 1462 | // A quadratic with start (0,0), control (0,10), end (10,10) elevates to a cubic whose controls |
| 1463 | // sit two-thirds of the way from each end towards (0,10): (0, 6.6667) and (3.3333, 10). |
| 1464 | let mut pb = PathBuilder::new(); |
| 1465 | pb.move_to(Pt::new(0.0, 0.0)); |
| 1466 | pb.quad_to(Pt::new(0.0, 10.0), Pt::new(10.0, 10.0)); |
| 1467 | let p = res!(pb.finish()); |
| 1468 | let mut s = String::new(); |
| 1469 | path_ops(&mut s, &p); |
| 1470 | // The move, then one cubic ending at the quadratic's endpoint. |
| 1471 | assert!(s.contains("0 0 m"), "the move, found: {}", s); |
| 1472 | assert!(s.contains(" c\n"), "a cubic operator, found: {}", s); |
| 1473 | assert!(s.contains("10 10 c"), "the cubic ends where the quadratic did, found: {}", s); |
| 1474 | Ok(()) |
| 1475 | } |
| 1476 | |
| 1477 | #[test] |
| 1478 | fn test_the_file_has_a_header_xref_and_trailer_01() -> Outcome<()> { |
| 1479 | let mut w = PdfWriter::new(); |
| 1480 | let mut page = PdfPage::new(100.0, 200.0); |
| 1481 | page.fill(res!(Path::rect(Bounds::new(10.0, 10.0, 90.0, 90.0))), Rgba::BLACK); |
| 1482 | w.add_page(page); |
| 1483 | let bytes = res!(w.to_bytes()); |
| 1484 | let text = String::from_utf8_lossy(&bytes); |
| 1485 | assert!(text.starts_with("%PDF-1.7"), "the header"); |
| 1486 | assert!(text.contains("/Type /Catalog"), "the catalogue"); |
| 1487 | assert!(text.contains("/MediaBox [0 0 100 200]"), "the media box, found in: {}", text); |
| 1488 | assert!(text.contains("1 0 0 -1 0 200 cm"), "the y-flip for a 200pt page"); |
| 1489 | assert!(text.contains("xref"), "the cross-reference table"); |
| 1490 | assert!(text.contains("startxref"), "the startxref"); |
| 1491 | assert!(text.trim_end().ends_with("%%EOF"), "the end-of-file marker"); |
| 1492 | Ok(()) |
| 1493 | } |
| 1494 | |
| 1495 | #[test] |
| 1496 | fn test_the_bytes_are_deterministic_02() -> Outcome<()> { |
| 1497 | // The same pages twice give the same bytes: no clock, no random source anywhere in the file. |
| 1498 | let build = || -> Outcome<Vec<u8>> { |
| 1499 | let mut w = PdfWriter::new(); |
| 1500 | let mut page = PdfPage::new(100.0, 100.0); |
| 1501 | page.fill(res!(Path::rect(Bounds::new(1.0, 1.0, 9.0, 9.0))), Rgba::new(10, 20, 30, 255)); |
| 1502 | w.add_page(page); |
| 1503 | w.to_bytes() |
| 1504 | }; |
| 1505 | assert_eq!(res!(build()), res!(build())); |
| 1506 | Ok(()) |
| 1507 | } |
| 1508 | |
| 1509 | #[test] |
| 1510 | fn test_the_outline_nests_by_level_04() -> Outcome<()> { |
| 1511 | // Two top-level entries, the second with a child and a grandchild, then a third top-level entry. |
| 1512 | let items = vec![ |
| 1513 | OutlineItem { title: "Title".into(), page: 0, level: 0 }, |
| 1514 | OutlineItem { title: "One".into(), page: 1, level: 0 }, |
| 1515 | OutlineItem { title: "One.a".into(), page: 2, level: 1 }, |
| 1516 | OutlineItem { title: "One.a.i".into(), page: 3, level: 2 }, |
| 1517 | OutlineItem { title: "Two".into(), page: 4, level: 0 }, |
| 1518 | ]; |
| 1519 | let (links, roots) = build_outline_links(&items); |
| 1520 | assert_eq!(roots, vec![0, 1, 4], "the three top-level entries are the roots"); |
| 1521 | // The first entry has no parent, no previous sibling, and "One" as its next. |
| 1522 | assert_eq!(links[0].parent, None); |
| 1523 | assert_eq!(links[0].prev, None); |
| 1524 | assert_eq!(links[0].next, Some(1)); |
| 1525 | // "One" parents "One.a", and its descendant count includes the grandchild. |
| 1526 | assert_eq!(links[1].first, Some(2)); |
| 1527 | assert_eq!(links[1].last, Some(2)); |
| 1528 | assert_eq!(links[1].count, 2, "child and grandchild are both descendants"); |
| 1529 | assert_eq!(links[1].next, Some(4), "Two is the next top-level sibling"); |
| 1530 | // "One.a" nests under "One" and parents the grandchild. |
| 1531 | assert_eq!(links[2].parent, Some(1)); |
| 1532 | assert_eq!(links[2].first, Some(3)); |
| 1533 | assert_eq!(links[3].parent, Some(2)); |
| 1534 | assert_eq!(links[3].count, 0, "the leaf has no descendants"); |
| 1535 | Ok(()) |
| 1536 | } |
| 1537 | |
| 1538 | #[test] |
| 1539 | fn test_the_outline_reaches_the_file_and_dest_pages_05() -> Outcome<()> { |
| 1540 | // A three-page document with a two-entry outline: the catalogue names the outline and the entries |
| 1541 | // carry destinations to their page objects (3 + 2*page). |
| 1542 | let mut w = PdfWriter::new(); |
| 1543 | for _ in 0..3 { |
| 1544 | let mut page = PdfPage::new(50.0, 50.0); |
| 1545 | page.fill(res!(Path::rect(Bounds::new(1.0, 1.0, 9.0, 9.0))), Rgba::BLACK); |
| 1546 | w.add_page(page); |
| 1547 | } |
| 1548 | w.set_outline(vec![ |
| 1549 | OutlineItem { title: "Title".into(), page: 0, level: 0 }, |
| 1550 | OutlineItem { title: "Body".into(), page: 2, level: 0 }, |
| 1551 | ]); |
| 1552 | let bytes = res!(w.to_bytes()); |
| 1553 | let text = String::from_utf8_lossy(&bytes); |
| 1554 | assert!(text.contains("/Outlines"), "the catalogue names an outline"); |
| 1555 | assert!(text.contains("/Type /Outlines"), "the outline root dict is present"); |
| 1556 | assert!(text.contains("/Title (Title)"), "the first entry's title"); |
| 1557 | assert!(text.contains("/Title (Body)"), "the second entry's title"); |
| 1558 | // Page 0 is object 3, page 2 is object 7. |
| 1559 | assert!(text.contains("/Dest [3 0 R /Fit]"), "the first entry jumps to page object 3"); |
| 1560 | assert!(text.contains("/Dest [7 0 R /Fit]"), "the second entry jumps to page object 7"); |
| 1561 | Ok(()) |
| 1562 | } |
| 1563 | |
| 1564 | #[test] |
| 1565 | fn test_no_outline_leaves_the_catalogue_untouched_06() -> Outcome<()> { |
| 1566 | // A document with no outline set carries the bare catalogue, byte for byte as before the feature. |
| 1567 | let mut w = PdfWriter::new(); |
| 1568 | let mut page = PdfPage::new(50.0, 50.0); |
| 1569 | page.fill(res!(Path::rect(Bounds::new(1.0, 1.0, 9.0, 9.0))), Rgba::BLACK); |
| 1570 | w.add_page(page); |
| 1571 | let bytes = res!(w.to_bytes()); |
| 1572 | let text = String::from_utf8_lossy(&bytes); |
| 1573 | assert!(text.contains("<< /Type /Catalog /Pages 2 0 R >>"), "the bare catalogue"); |
| 1574 | assert!(!text.contains("/Outlines"), "no outline object when none was set"); |
| 1575 | Ok(()) |
| 1576 | } |
| 1577 | |
| 1578 | #[test] |
| 1579 | fn test_a_non_ascii_title_encodes_as_utf16_07() -> Outcome<()> { |
| 1580 | // A title with an em dash cannot be a printable-ASCII literal, so it is a UTF-16BE hex string with |
| 1581 | // a byte-order mark. |
| 1582 | let s = pdf_text_string("A — B"); |
| 1583 | assert!(s.starts_with("<FEFF"), "a UTF-16BE hex string, found: {}", s); |
| 1584 | assert!(s.ends_with('>'), "closed hex string"); |
| 1585 | // A plain title stays a readable literal. |
| 1586 | assert_eq!(pdf_text_string("Contents"), "(Contents)"); |
| 1587 | // Parentheses and backslashes in a literal are escaped. |
| 1588 | assert_eq!(pdf_text_string("a (b) \\ c"), "(a \\(b\\) \\\\ c)"); |
| 1589 | Ok(()) |
| 1590 | } |
| 1591 | |
| 1592 | #[test] |
| 1593 | fn test_a_link_annotation_is_emitted_over_a_rect_08() -> Outcome<()> { |
| 1594 | // A page 100 wide by 200 tall with a link over the rectangle at top-left (10, 20), 30 wide, 40 tall. |
| 1595 | // The page dict names an /Annots array, and the annotation is a /Link with a URI action whose /Rect |
| 1596 | // is the rectangle flipped into PDF's y-up frame: [10, 200-60, 40, 200-20] = [10 140 40 180]. |
| 1597 | let mut w = PdfWriter::new(); |
| 1598 | let mut page = PdfPage::new(100.0, 200.0); |
| 1599 | page.fill(res!(Path::rect(Bounds::new(10.0, 10.0, 90.0, 90.0))), Rgba::BLACK); |
| 1600 | page.link(10.0, 20.0, 30.0, 40.0, "https://need2know.ai/with-ai/doc".to_string()); |
| 1601 | w.add_page(page); |
| 1602 | let bytes = res!(w.to_bytes()); |
| 1603 | let text = String::from_utf8_lossy(&bytes); |
| 1604 | assert!(text.contains("/Annots"), "the page dict names an annotation array, found: {}", text); |
| 1605 | assert!(text.contains("/Subtype /Link"), "a link annotation is written"); |
| 1606 | assert!(text.contains("/S /URI /URI (https://need2know.ai/with-ai/doc)"), "the URI action, found: {}", text); |
| 1607 | assert!(text.contains("/Rect [10 140 40 180]"), "the rectangle flipped into PDF space, found: {}", text); |
| 1608 | assert!(text.contains("/Border [0 0 0]"), "the border is drawn away"); |
| 1609 | Ok(()) |
| 1610 | } |
| 1611 | |
| 1612 | #[test] |
| 1613 | fn test_no_link_leaves_the_page_bytes_identical_09() -> Outcome<()> { |
| 1614 | // A page with no link writes no /Annots and no annotation object -- byte for byte as before the |
| 1615 | // feature. The two builds of the same annot-free page also agree, so the field adds no nondeterminism. |
| 1616 | let build = || -> Outcome<Vec<u8>> { |
| 1617 | let mut w = PdfWriter::new(); |
| 1618 | let mut page = PdfPage::new(100.0, 100.0); |
| 1619 | page.fill(res!(Path::rect(Bounds::new(1.0, 1.0, 9.0, 9.0))), Rgba::new(10, 20, 30, 255)); |
| 1620 | w.add_page(page); |
| 1621 | w.to_bytes() |
| 1622 | }; |
| 1623 | let bytes = res!(build()); |
| 1624 | let text = String::from_utf8_lossy(&bytes); |
| 1625 | assert!(!text.contains("/Annots"), "no annotation array when the page carries no link"); |
| 1626 | assert!(!text.contains("/Subtype /Link"), "no link annotation is written"); |
| 1627 | assert_eq!(res!(build()), bytes, "an annot-free page is deterministic"); |
| 1628 | Ok(()) |
| 1629 | } |
| 1630 | |
| 1631 | #[test] |
| 1632 | fn test_a_coordinate_rounds_to_three_places_10() -> Outcome<()> { |
| 1633 | // Full f32 precision is spent on noise below a thousandth of a point; three places is far finer |
| 1634 | // than any renderer resolves. A rounded -0 folds back to 0 so the bytes stay canonical. |
| 1635 | assert_eq!(numf32(841.8897705078125), "841.89"); |
| 1636 | assert_eq!(numf(3.14159), "3.142"); |
| 1637 | assert_eq!(numf(0.0004), "0"); |
| 1638 | assert_eq!(numf(-0.0004), "0"); |
| 1639 | assert_eq!(numf(12.5), "12.5"); |
| 1640 | assert_eq!(numf(100.0), "100"); |
| 1641 | Ok(()) |
| 1642 | } |
| 1643 | |
| 1644 | #[test] |
| 1645 | fn test_compression_shrinks_and_flate_decodes_11() -> Outcome<()> { |
| 1646 | use flate2::read::ZlibDecoder; |
| 1647 | use std::io::Read; |
| 1648 | |
| 1649 | // The same page compressed and uncompressed: the compressed file names /FlateDecode on its content |
| 1650 | // stream, is smaller, and its stream inflates back to the operators the uncompressed file writes in |
| 1651 | // the clear. |
| 1652 | let build = |compress: bool| -> Outcome<Vec<u8>> { |
| 1653 | let mut w = PdfWriter::new().with_compression(compress); |
| 1654 | let mut page = PdfPage::new(200.0, 200.0); |
| 1655 | for i in 0..200 { |
| 1656 | let o = i as f32 * 0.1; |
| 1657 | page.fill(res!(Path::rect(Bounds::new(1.0 + o, 1.0 + o, 9.0 + o, 9.0 + o))), Rgba::BLACK); |
| 1658 | } |
| 1659 | w.add_page(page); |
| 1660 | w.to_bytes() |
| 1661 | }; |
| 1662 | let plain = res!(build(false)); |
| 1663 | let zipped = res!(build(true)); |
| 1664 | assert!(zipped.len() < plain.len(), |
| 1665 | "compression shrank the file: {} < {}", zipped.len(), plain.len()); |
| 1666 | let ztext = String::from_utf8_lossy(&zipped); |
| 1667 | assert!(ztext.contains("/Filter /FlateDecode"), "the content stream is flate-filtered"); |
| 1668 | |
| 1669 | // The content object is object 4 (catalogue, page tree, page, content); pull its stream bytes and |
| 1670 | // inflate them. |
| 1671 | let obj = b"4 0 obj"; |
| 1672 | let at = match zipped.windows(obj.len()).position(|w| w == obj) { |
| 1673 | Some(i) => i, |
| 1674 | None => return Err(err!("no content object in the compressed file"; Test)), |
| 1675 | }; |
| 1676 | let tail = &zipped[at..]; |
| 1677 | let sopen = b"stream\n"; |
| 1678 | let sp = match tail.windows(sopen.len()).position(|w| w == sopen) { |
| 1679 | Some(i) => i + sopen.len(), |
| 1680 | None => return Err(err!("the content object opens no stream"; Test)), |
| 1681 | }; |
| 1682 | let sclose = b"\nendstream"; |
| 1683 | let ep = match tail[sp..].windows(sclose.len()).position(|w| w == sclose) { |
| 1684 | Some(i) => sp + i, |
| 1685 | None => return Err(err!("the content stream is unterminated"; Test)), |
| 1686 | }; |
| 1687 | let mut dec = ZlibDecoder::new(&tail[sp..ep]); |
| 1688 | let mut raw = Vec::new(); |
| 1689 | res!(dec.read_to_end(&mut raw)); |
| 1690 | let ctext = String::from_utf8_lossy(&raw); |
| 1691 | assert!(ctext.contains("1 0 0 -1 0 200 cm"), "the inflated stream holds the page flip"); |
| 1692 | assert!(ctext.contains("\nf\n"), "the inflated stream holds fill operators"); |
| 1693 | Ok(()) |
| 1694 | } |
| 1695 | |
| 1696 | /// A small closed outline for glyph tests, offset by `dx` so two calls make two distinct outlines. |
| 1697 | fn tri(dx: f32) -> Outcome<Path> { |
| 1698 | let mut pb = PathBuilder::new(); |
| 1699 | pb.move_to(Pt::new(dx, 0.0)); |
| 1700 | pb.line_to(Pt::new(dx + 100.0, 0.0)); |
| 1701 | pb.line_to(Pt::new(dx + 50.0, 200.0)); |
| 1702 | pb.close(); |
| 1703 | pb.finish() |
| 1704 | } |
| 1705 | |
| 1706 | #[test] |
| 1707 | fn test_a_repeated_glyph_makes_one_charproc_12() -> Outcome<()> { |
| 1708 | // A glyph drawn twice on a page is stored once as a Type-3 CharProc and shown twice by one code; a |
| 1709 | // second, different outline adds a second CharProc. Text is shown with text operators, not inline |
| 1710 | // path fills. Built uncompressed so the structure is readable in the bytes. |
| 1711 | let mut w = PdfWriter::new(); |
| 1712 | let mut page = PdfPage::new(200.0, 200.0); |
| 1713 | let a = res!(tri(0.0)); |
| 1714 | let b = res!(tri(100.0)); |
| 1715 | page.glyph(a.clone(), 10.0, 50.0, 12.0, 8.0, Rgba::BLACK, "A".into()); |
| 1716 | page.glyph(a, 30.0, 50.0, 12.0, 8.0, Rgba::BLACK, "A".into()); |
| 1717 | page.glyph(b, 50.0, 50.0, 12.0, 8.0, Rgba::BLACK, "B".into()); |
| 1718 | w.add_page(page); |
| 1719 | let bytes = res!(w.to_bytes()); |
| 1720 | let text = String::from_utf8_lossy(&bytes); |
| 1721 | |
| 1722 | assert!(text.contains("/Subtype /Type3"), "a Type-3 font is emitted"); |
| 1723 | assert!(text.contains("BT\n"), "a text object opens"); |
| 1724 | assert!(text.contains(" Tj\n"), "glyphs are shown with Tj, not inline fills"); |
| 1725 | assert!(text.contains("/ToUnicode"), "a ToUnicode CMap is referenced"); |
| 1726 | // Two distinct glyphs give /g0 and /g1 and no /g2 -- the repeat added no CharProc. |
| 1727 | assert!(text.contains("/g0 "), "the first glyph's CharProc"); |
| 1728 | assert!(text.contains("/g1 "), "the second glyph's CharProc"); |
| 1729 | assert!(!text.contains("/g2"), "the repeated glyph added no third CharProc"); |
| 1730 | // The repeated glyph is code 0, shown twice; the other is code 1, shown once. |
| 1731 | assert_eq!(text.matches("<00> Tj").count(), 2, "the repeat shows one code twice"); |
| 1732 | assert_eq!(text.matches("<01> Tj").count(), 1, "the other glyph shows once"); |
| 1733 | Ok(()) |
| 1734 | } |
| 1735 | |
| 1736 | #[test] |
| 1737 | fn test_glyph_bytes_are_deterministic_13() -> Outcome<()> { |
| 1738 | // The same glyphs twice give the same bytes: code assignment is by draw order, and the font tables |
| 1739 | // serialise in a fixed order, so nothing depends on hash-map iteration. |
| 1740 | let build = || -> Outcome<Vec<u8>> { |
| 1741 | let mut w = PdfWriter::new().with_compression(true); |
| 1742 | let mut page = PdfPage::new(200.0, 200.0); |
| 1743 | let a = res!(tri(0.0)); |
| 1744 | let b = res!(tri(100.0)); |
| 1745 | page.glyph(a.clone(), 10.0, 50.0, 12.0, 8.0, Rgba::BLACK, "A".into()); |
| 1746 | page.glyph(b, 30.0, 50.0, 12.0, 8.0, Rgba::BLACK, "B".into()); |
| 1747 | page.glyph(a, 50.0, 50.0, 12.0, 8.0, Rgba::BLACK, "A".into()); |
| 1748 | w.add_page(page); |
| 1749 | w.to_bytes() |
| 1750 | }; |
| 1751 | assert_eq!(res!(build()), res!(build())); |
| 1752 | Ok(()) |
| 1753 | } |
| 1754 | |
| 1755 | #[test] |
| 1756 | fn test_tounicode_maps_codes_to_source_text_14() -> Outcome<()> { |
| 1757 | // The /ToUnicode CMap maps each code to the UTF-16BE of its source text, so a viewer extracts the |
| 1758 | // real characters rather than the font's private codes. Built uncompressed so the CMap is readable. |
| 1759 | let mut w = PdfWriter::new(); |
| 1760 | let mut page = PdfPage::new(200.0, 200.0); |
| 1761 | page.glyph(res!(tri(0.0)), 10.0, 50.0, 12.0, 8.0, Rgba::BLACK, "O".into()); |
| 1762 | page.glyph(res!(tri(100.0)), 30.0, 50.0, 12.0, 8.0, Rgba::BLACK, "x".into()); |
| 1763 | w.add_page(page); |
| 1764 | let bytes = res!(w.to_bytes()); |
| 1765 | let text = String::from_utf8_lossy(&bytes); |
| 1766 | assert!(text.contains("beginbfchar"), "the CMap carries character mappings"); |
| 1767 | assert!(text.contains("<00> <004F>"), "code 0 maps to 'O' (U+004F)"); |
| 1768 | assert!(text.contains("<01> <0078>"), "code 1 maps to 'x' (U+0078)"); |
| 1769 | Ok(()) |
| 1770 | } |
| 1771 | |
| 1772 | /// Every xref entry of `bytes` points at the object it names. |
| 1773 | fn xref_lands(bytes: &[u8]) -> Outcome<usize> { |
| 1774 | let needle = b"xref\n0 "; |
| 1775 | let marker = match bytes.windows(needle.len()).position(|w| w == needle) { |
| 1776 | Some(i) => i, |
| 1777 | None => return Err(err!("no xref section in the file"; Test)), |
| 1778 | }; |
| 1779 | let nl2 = match bytes[marker + needle.len()..].iter().position(|&b| b == b'\n') { |
| 1780 | Some(i) => marker + needle.len() + i, |
| 1781 | None => return Err(err!("the xref header is malformed"; Test)), |
| 1782 | }; |
| 1783 | let count = res!(res!(std::str::from_utf8(&bytes[marker + needle.len()..nl2])).trim().parse::<usize>()); |
| 1784 | let entries = &bytes[nl2 + 1..]; |
| 1785 | for obj in 1..count { |
| 1786 | let off = res!(res!(std::str::from_utf8(&entries[obj * 20..obj * 20 + 10])).parse::<usize>()); |
| 1787 | let want = fmt!("{} 0 obj", obj); |
| 1788 | assert!(bytes[off..].starts_with(want.as_bytes()), "object {} is not at offset {}", obj, off); |
| 1789 | } |
| 1790 | Ok(count - 1) |
| 1791 | } |
| 1792 | |
| 1793 | #[test] |
| 1794 | fn test_embedded_text_runs_as_one_tj_with_exact_adjustments_15() -> Outcome<()> { |
| 1795 | // Three glyphs of a TrueType face on one baseline: the second sits exactly where the first's advance |
| 1796 | // leaves the pen, the third two points further. One text matrix, one TJ, and a single adjustment of |
| 1797 | // -2pt at 12pt, which is -2 * 1000 / 12 = -167 thousandths of the em. |
| 1798 | let data = Arc::new(include_bytes!("../../fe2o3_font/fonts/DejaVuSans.ttf").to_vec()); |
| 1799 | let prog = match res!(FontProgram::parse(data)) { |
| 1800 | Some(p) => Arc::new(p), |
| 1801 | None => return Err(err!("DejaVu Sans should be embeddable"; Test)), |
| 1802 | }; |
| 1803 | let (a, b, c) = (36u16, 37u16, 38u16); |
| 1804 | let xa = 10.0f32; |
| 1805 | let xb = xa + prog.width(a) as f32 * 12.0 / 1000.0; |
| 1806 | let xc = xb + prog.width(b) as f32 * 12.0 / 1000.0 + 2.0; |
| 1807 | let mut page = PdfPage::new(200.0, 200.0); |
| 1808 | for (g, x, t) in [(a, xa, "A"), (b, xb, "B"), (c, xc, "C")] { |
| 1809 | page.text(prog.clone(), g, x, 50.0, 12.0, Rgba::BLACK, t.into()); |
| 1810 | } |
| 1811 | let mut w = PdfWriter::new(); |
| 1812 | w.add_page(page); |
| 1813 | let bytes = res!(w.to_bytes()); |
| 1814 | let text = String::from_utf8_lossy(&bytes); |
| 1815 | assert_eq!(text.matches(" Tm\n").count(), 1, "one text matrix for the run"); |
| 1816 | assert!(text.contains("[<00240025>-167<0026>] TJ"), "one TJ with one adjustment, found: {}", text); |
| 1817 | assert!(text.contains("/Subtype /Type0"), "a composite font"); |
| 1818 | assert!(text.contains("/Subtype /CIDFontType2"), "over a TrueType CIDFont"); |
| 1819 | assert!(text.contains("/CIDToGIDMap /Identity"), "whose CIDs are its glyph ids"); |
| 1820 | assert!(text.contains(&fmt!("/W [36 [{} {} {}]]", prog.width(a), prog.width(b), prog.width(c))), |
| 1821 | "the widths of the three glyphs, found: {}", text); |
| 1822 | assert!(text.contains("<0024> <0041>"), "the ToUnicode maps glyph 36 to 'A'"); |
| 1823 | assert!(res!(xref_lands(&bytes)) > 8, "the font objects are in the xref"); |
| 1824 | Ok(()) |
| 1825 | } |
| 1826 | |
| 1827 | #[test] |
| 1828 | fn test_the_xref_offsets_land_on_their_objects_03() -> Outcome<()> { |
| 1829 | // The heart of a valid PDF: every offset in the cross-reference table must point at the first |
| 1830 | // byte of the object it names. This reads each twenty-byte entry's offset back and confirms the |
| 1831 | // object at that offset opens with "N 0 obj", which catches an off-by-one in the byte accounting. |
| 1832 | // The page carries glyphs as well as a fill, so the CharProc, ToUnicode and font-dictionary objects |
| 1833 | // appended at finish() are covered too -- the object count is read from the xref header, not assumed. |
| 1834 | let mut w = PdfWriter::new(); |
| 1835 | let mut page = PdfPage::new(50.0, 50.0); |
| 1836 | page.fill(res!(Path::rect(Bounds::new(1.0, 1.0, 9.0, 9.0))), Rgba::BLACK); |
| 1837 | page.glyph(res!(tri(0.0)), 10.0, 20.0, 12.0, 8.0, Rgba::BLACK, "O".into()); |
| 1838 | page.glyph(res!(tri(100.0)), 20.0, 20.0, 12.0, 8.0, Rgba::BLACK, "x".into()); |
| 1839 | w.add_page(page); |
| 1840 | let bytes = res!(w.to_bytes()); |
| 1841 | |
| 1842 | // The entries begin after "xref\n" and the "0 M\n" subsection header, where M is the object count |
| 1843 | // plus the free entry. The free object is entry zero; objects 1..=obj_count follow, twenty bytes |
| 1844 | // each. |
| 1845 | // Search the raw bytes, not a lossy string: the header's binary-marker comment holds non-UTF-8 |
| 1846 | // bytes, so a String index would not line up with the byte offsets the entries are read at. |
| 1847 | let needle = b"xref\n0 "; |
| 1848 | let marker = match bytes.windows(needle.len()).position(|w| w == needle) { |
| 1849 | Some(i) => i, |
| 1850 | None => return Err(err!("no xref section in the file"; Test)), |
| 1851 | }; |
| 1852 | let nl1 = match bytes[marker..].iter().position(|&b| b == b'\n') { |
| 1853 | Some(i) => marker + i, |
| 1854 | None => return Err(err!("the xref header is malformed"; Test)), |
| 1855 | }; |
| 1856 | let nl2 = match bytes[nl1 + 1..].iter().position(|&b| b == b'\n') { |
| 1857 | Some(i) => nl1 + 1 + i, |
| 1858 | None => return Err(err!("the xref header is malformed"; Test)), |
| 1859 | }; |
| 1860 | // The subsection header reads "0 {count}", the count being every object plus the free entry, so the |
| 1861 | // highest numbered object is one less. Its digits run from just past the needle (the newline nl1 sits |
| 1862 | // inside "xref\n", before them) to nl2. Reading it here covers the font objects appended at finish(). |
| 1863 | let count_str = res!(std::str::from_utf8(&bytes[marker + needle.len()..nl2])); |
| 1864 | let obj_count = res!(count_str.trim().parse::<usize>()) - 1; |
| 1865 | assert!(obj_count > 4, "the glyphs added font objects past the four base objects: {}", obj_count); |
| 1866 | let entries = &bytes[nl2 + 1..]; |
| 1867 | for obj in 1..=obj_count { |
| 1868 | let field = res!(std::str::from_utf8(&entries[obj * 20..obj * 20 + 10])); |
| 1869 | let off: usize = res!(field.parse::<usize>()); |
| 1870 | let want = fmt!("{} 0 obj", obj); |
| 1871 | assert!(bytes[off..].starts_with(want.as_bytes()), |
| 1872 | "object {} offset {} does not open with '{}'", obj, off, want); |
| 1873 | } |
| 1874 | Ok(()) |
| 1875 | } |
| 1876 | } |