Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_graphics/src/pdf_font.rs

35.2 KiB, 1 run

created by r1870400018:59746, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! The font programs a PDF embeds, and their subsets.
2//!
3//! A [`FontProgram`] is an OpenType or TrueType file read just far enough to be embedded: its outline
4//! flavour, its metrics for the font descriptor, its advance widths, and its PostScript name. The
5//! [`subset`](FontProgram::subset) keeps only the glyphs a document shows, yet a glyph id from the shaper
6//! is the PDF's CID unchanged either way, so no remapping table sits between the page and the font.
7//!
8//! TrueType (`glyf`) outlines become a trimmed `sfnt` for `/FontFile2`, the unused glyphs emptied in
9//! place. CFF outlines become a bare CID-keyed CFF for `/FontFile3 /CIDFontType0C`, the kept glyphs packed
10//! with their subroutines inlined and the charset mapping each back to its original id as its CID. A
11//! name-keyed font is re-keyed this way too, because a name-keyed program inside a CID font is read
12//! inconsistently across viewers. A `CFF2` (variable) font and a font whose licence forbids embedding
13//! are refused with `None`, and the caller draws that face's glyphs as outlines instead.
14
15use oxedyne_fe2o3_core::prelude::*;
16
17use std::collections::BTreeSet;
18use std::sync::Arc;
19
20/// The outline flavour of an embeddable program, which decides the PDF font subtype.
21#[derive(Clone, Copy, Debug, PartialEq, Eq)]
22pub enum Outlines {
23 TrueType, // `glyf`, embedded as CIDFontType2 over /FontFile2
24 Cff, // `CFF `, embedded as CIDFontType0 over /FontFile3 /CIDFontType0C
25}
26
27/// A font program ready for a PDF font file stream, in the form its `/FontFile` key and subtype need.
28#[derive(Clone, Debug)]
29pub enum FontFile {
30 TrueType(Vec<u8>), // /FontFile2
31 Cff(Vec<u8>), // /FontFile3 /Subtype /CIDFontType0C
32 OpenType(Vec<u8>), // /FontFile3 /Subtype /OpenType, the whole file when a CFF subset fails
33}
34
35/// One embeddable font file, parsed once and shared. The bytes are held whole; only
36/// [`subset`](Self::subset) cuts them down, at the end of a document when the glyphs shown are known.
37pub struct FontProgram {
38 key: u64, // content fingerprint, so two loads of one file are one PDF font
39 data: Arc<Vec<u8>>,
40 outlines: Outlines,
41 tables: Vec<Table>,
42 upem: u16,
43 advances: Vec<u16>, // per glyph, font units
44 num_glyphs: u16,
45 bbox: [i16; 4], // head xMin, yMin, xMax, yMax
46 ascent: i16,
47 descent: i16,
48 cap_height: i16,
49 weight: u16,
50 italic_angle: f32,
51 fixed_pitch: bool,
52 may_subset: bool, // OS/2 fsType does not forbid subsetting
53 name: String, // PostScript name, sanitised for a PDF name object
54}
55
56impl std::fmt::Debug for FontProgram {
57 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
58 f.debug_struct("FontProgram")
59 .field("name", &self.name)
60 .field("outlines", &self.outlines)
61 .field("glyphs", &self.num_glyphs)
62 .field("bytes", &self.data.len())
63 .finish()
64 }
65}
66
67#[derive(Clone, Copy, Debug)]
68struct Table {
69 tag: [u8; 4],
70 off: usize,
71 len: usize,
72}
73
74/// A 64-bit FNV-1a over the whole file.
75fn fingerprint(bytes: &[u8]) -> u64 {
76 let mut h: u64 = 0xcbf2_9ce4_8422_2325;
77 for &b in bytes {
78 h ^= b as u64;
79 h = h.wrapping_mul(0x0000_0100_0000_01b3);
80 }
81 h
82}
83
84fn rd_u8(d: &[u8], at: usize) -> Outcome<u8> {
85 match d.get(at) {
86 Some(b) => Ok(*b),
87 None => Err(err!("A font read at byte {} ran past the end of {} bytes.", at, d.len();
88 Invalid, Input, Size)),
89 }
90}
91
92fn rd_u16(d: &[u8], at: usize) -> Outcome<u16> {
93 Ok(((res!(rd_u8(d, at)) as u16) << 8) | res!(rd_u8(d, at + 1)) as u16)
94}
95
96fn rd_i16(d: &[u8], at: usize) -> Outcome<i16> {
97 Ok(res!(rd_u16(d, at)) as i16)
98}
99
100fn rd_u32(d: &[u8], at: usize) -> Outcome<u32> {
101 Ok(((res!(rd_u16(d, at)) as u32) << 16) | res!(rd_u16(d, at + 2)) as u32)
102}
103
104/// An unsigned big-endian integer of `n` bytes, one to four, as CFF offsets are written.
105fn rd_uint(d: &[u8], at: usize, n: usize) -> Outcome<usize> {
106 let mut v = 0usize;
107 for k in 0..n {
108 v = (v << 8) | res!(rd_u8(d, at + k)) as usize;
109 }
110 Ok(v)
111}
112
113fn slice(d: &[u8], at: usize, len: usize) -> Outcome<&[u8]> {
114 match at.checked_add(len).and_then(|end| d.get(at..end)) {
115 Some(s) => Ok(s),
116 None => Err(err!("A font span {}+{} runs past the end of {} bytes.", at, len, d.len();
117 Invalid, Input, Size)),
118 }
119}
120
121impl FontProgram {
122
123 /// Reads a font file for embedding. `None` when the file is well formed but cannot be embedded: a
124 /// `CFF2` variable font, a bitmap-only licence, or a restricted-licence font whose `fsType` forbids
125 /// embedding altogether. An error only when the file is not a font this reader understands.
126 pub fn parse(data: Arc<Vec<u8>>) -> Outcome<Option<Self>> {
127 let d = &data[..];
128 let key = fingerprint(d);
129 let version = res!(rd_u32(d, 0));
130 if version == 0x7474_6366 { // 'ttcf', a collection: its faces arrive one by one elsewhere
131 return Ok(None);
132 }
133 let count = res!(rd_u16(d, 4)) as usize;
134 let mut tables = Vec::with_capacity(count);
135 for i in 0..count {
136 let rec = 12 + 16 * i;
137 let tag = res!(slice(d, rec, 4));
138 let off = res!(rd_u32(d, rec + 8)) as usize;
139 let len = res!(rd_u32(d, rec + 12)) as usize;
140 res!(slice(d, off, len));
141 tables.push(Table { tag: [tag[0], tag[1], tag[2], tag[3]], off, len });
142 }
143 let find = |t: &[u8; 4]| tables.iter().find(|x| &x.tag == t).copied();
144
145 let outlines = if find(b"glyf").is_some() && find(b"loca").is_some() {
146 Outlines::TrueType
147 } else if find(b"CFF ").is_some() {
148 Outlines::Cff
149 } else {
150 return Ok(None); // CFF2, or bitmap-only
151 };
152
153 let head = match find(b"head") {
154 Some(t) => t,
155 None => return Err(err!("The font has no head table."; Invalid, Input, Missing)),
156 };
157 let hhea = match find(b"hhea") {
158 Some(t) => t,
159 None => return Err(err!("The font has no hhea table."; Invalid, Input, Missing)),
160 };
161 let maxp = match find(b"maxp") {
162 Some(t) => t,
163 None => return Err(err!("The font has no maxp table."; Invalid, Input, Missing)),
164 };
165 let hmtx = match find(b"hmtx") {
166 Some(t) => t,
167 None => return Err(err!("The font has no hmtx table."; Invalid, Input, Missing)),
168 };
169
170 let upem = res!(rd_u16(d, head.off + 18));
171 if upem == 0 {
172 return Err(err!("The font declares zero units per em."; Invalid, Input));
173 }
174 let bbox = [
175 res!(rd_i16(d, head.off + 36)),
176 res!(rd_i16(d, head.off + 38)),
177 res!(rd_i16(d, head.off + 40)),
178 res!(rd_i16(d, head.off + 42)),
179 ];
180 let mut ascent = res!(rd_i16(d, hhea.off + 4));
181 let mut descent = res!(rd_i16(d, hhea.off + 6));
182 let n_hm = res!(rd_u16(d, hhea.off + 34)) as usize;
183 let num_glyphs = res!(rd_u16(d, maxp.off + 4));
184
185 // Advances: one long metric per glyph up to numberOfHMetrics, the last repeated after.
186 let mut advances = Vec::with_capacity(num_glyphs as usize);
187 let mut last = 0u16;
188 for g in 0..num_glyphs as usize {
189 if g < n_hm {
190 last = res!(rd_u16(d, hmtx.off + 4 * g));
191 }
192 advances.push(last);
193 }
194
195 let mut cap_height = bbox[3];
196 let mut weight = 400u16;
197 let mut may_subset = true;
198 if let Some(os2) = find(b"OS/2") {
199 let ver = res!(rd_u16(d, os2.off));
200 weight = res!(rd_u16(d, os2.off + 4));
201 let fs_type = res!(rd_u16(d, os2.off + 8));
202 // Bit 1 alone is a restricted licence; bit 9 permits bitmaps only. Either rules out an outline.
203 if fs_type & 0x000f == 0x0002 || fs_type & 0x0200 != 0 {
204 return Ok(None);
205 }
206 may_subset = fs_type & 0x0100 == 0;
207 if os2.len >= 72 {
208 let typo_asc = res!(rd_i16(d, os2.off + 68));
209 let typo_des = res!(rd_i16(d, os2.off + 70));
210 if typo_asc != 0 { ascent = typo_asc; }
211 if typo_des != 0 { descent = typo_des; }
212 }
213 if ver >= 2 && os2.len >= 90 {
214 let ch = res!(rd_i16(d, os2.off + 88));
215 if ch > 0 { cap_height = ch; }
216 }
217 }
218
219 let mut italic_angle = 0.0f32;
220 let mut fixed_pitch = false;
221 if let Some(post) = find(b"post") {
222 italic_angle = (res!(rd_u32(d, post.off + 4)) as i32) as f32 / 65536.0;
223 fixed_pitch = res!(rd_u32(d, post.off + 12)) != 0;
224 }
225
226 let name = match find(b"name") {
227 Some(t) => postscript_name(d, t).unwrap_or_default(),
228 None => String::new(),
229 };
230 let name = if name.is_empty() { "Font".to_string() } else { name };
231
232 Ok(Some(Self {
233 key,
234 data,
235 outlines,
236 tables,
237 upem,
238 advances,
239 num_glyphs,
240 bbox,
241 ascent,
242 descent,
243 cap_height,
244 weight,
245 italic_angle,
246 fixed_pitch,
247 may_subset,
248 name,
249 }))
250 }
251
252 pub fn key(&self) -> u64 { self.key }
253 pub fn outlines(&self) -> Outlines { self.outlines }
254 pub fn name(&self) -> &str { &self.name }
255 pub fn num_glyphs(&self) -> u16 { self.num_glyphs }
256
257 /// A glyph's advance in thousandths of an em, the unit of a PDF `/W` array and a `TJ` adjustment.
258 pub fn width(&self, gid: u16) -> i64 {
259 let adv = self.advances.get(gid as usize).copied().unwrap_or(0) as f64;
260 (adv * 1000.0 / self.upem as f64).round() as i64
261 }
262
263 /// A font-unit value in thousandths of an em.
264 pub(crate) fn em(&self, v: i16) -> i64 {
265 ((v as f64) * 1000.0 / self.upem as f64).round() as i64
266 }
267
268 pub(crate) fn bbox(&self) -> [i64; 4] {
269 [self.em(self.bbox[0]), self.em(self.bbox[1]), self.em(self.bbox[2]), self.em(self.bbox[3])]
270 }
271 pub(crate) fn ascent(&self) -> i64 { self.em(self.ascent) }
272 pub(crate) fn descent(&self) -> i64 { self.em(self.descent) }
273 pub(crate) fn cap_height(&self) -> i64 { self.em(self.cap_height) }
274 pub(crate) fn italic_angle(&self) -> f32 { self.italic_angle }
275 pub(crate) fn fixed_pitch(&self) -> bool { self.fixed_pitch }
276 pub(crate) fn weight(&self) -> u16 { self.weight }
277
278 fn table(&self, tag: &[u8; 4]) -> Option<&[u8]> {
279 self.tables.iter()
280 .find(|t| &t.tag == tag)
281 .and_then(|t| self.data.get(t.off..t.off + t.len))
282 }
283
284 /// The font program cut down to `gids` (glyph zero is always kept). A font whose licence forbids
285 /// subsetting keeps every glyph. Should a subset fail on a construction this reader does not follow --
286 /// a `seac` accent, a malformed subroutine -- the whole file is embedded instead, which costs bytes
287 /// but still shows every glyph by its id.
288 pub fn subset(&self, gids: &BTreeSet<u16>) -> FontFile {
289 match self.try_subset(gids) {
290 Ok(f) => f,
291 Err(_) => match self.outlines {
292 Outlines::TrueType => FontFile::TrueType(self.data.to_vec()),
293 Outlines::Cff => FontFile::OpenType(self.data.to_vec()),
294 },
295 }
296 }
297
298 fn try_subset(&self, gids: &BTreeSet<u16>) -> Outcome<FontFile> {
299 let mut keep: BTreeSet<u16> = if self.may_subset {
300 gids.iter().copied().filter(|&g| g < self.num_glyphs).collect()
301 } else {
302 (0..self.num_glyphs).collect()
303 };
304 keep.insert(0);
305 match self.outlines {
306 Outlines::TrueType => Ok(FontFile::TrueType(res!(self.subset_truetype(keep)))),
307 Outlines::Cff => {
308 let cff = match self.table(b"CFF ") {
309 Some(t) => t,
310 None => return Err(err!("The CFF table vanished between parse and subset."; Bug)),
311 };
312 Ok(FontFile::Cff(res!(subset_cff(cff, &keep, self.num_glyphs as usize))))
313 },
314 }
315 }
316
317 // ┌───────────────────────────────────────────────────────────────────────────┐
318 // │ TRUETYPE │
319 // └───────────────────────────────────────────────────────────────────────────┘
320
321 fn subset_truetype(&self, mut keep: BTreeSet<u16>) -> Outcome<Vec<u8>> {
322 let head = match self.table(b"head") {
323 Some(t) => t,
324 None => return Err(err!("The font has no head table."; Invalid, Input, Missing)),
325 };
326 let loca = match self.table(b"loca") {
327 Some(t) => t,
328 None => return Err(err!("The font has no loca table."; Invalid, Input, Missing)),
329 };
330 let glyf = match self.table(b"glyf") {
331 Some(t) => t,
332 None => return Err(err!("The font has no glyf table."; Invalid, Input, Missing)),
333 };
334 let long = res!(rd_i16(head, 50)) != 0;
335 let n = self.num_glyphs as usize;
336 let loc = |g: usize| -> Outcome<usize> {
337 if long {
338 Ok(res!(rd_u32(loca, 4 * g)) as usize)
339 } else {
340 Ok(res!(rd_u16(loca, 2 * g)) as usize * 2)
341 }
342 };
343 let glyph = |g: usize| -> Outcome<&[u8]> {
344 let a = res!(loc(g));
345 let b = res!(loc(g + 1));
346 if b < a {
347 return Err(err!("Glyph {} has a negative length in loca.", g; Invalid, Input));
348 }
349 slice(glyf, a, b - a)
350 };
351
352 // A composite glyph draws other glyphs, which must survive the subset with it.
353 let mut todo: Vec<u16> = keep.iter().copied().collect();
354 while let Some(g) = todo.pop() {
355 let bytes = res!(glyph(g as usize));
356 if bytes.len() < 10 || res!(rd_i16(bytes, 0)) >= 0 {
357 continue;
358 }
359 let mut at = 10;
360 loop {
361 let flags = res!(rd_u16(bytes, at));
362 let comp = res!(rd_u16(bytes, at + 2));
363 if (comp as usize) < n && keep.insert(comp) {
364 todo.push(comp);
365 }
366 at += 4;
367 at += if flags & 0x0001 != 0 { 4 } else { 2 }; // ARG_1_AND_2_ARE_WORDS
368 if flags & 0x0008 != 0 { // WE_HAVE_A_SCALE
369 at += 2;
370 } else if flags & 0x0040 != 0 { // WE_HAVE_AN_X_AND_Y_SCALE
371 at += 4;
372 } else if flags & 0x0080 != 0 { // WE_HAVE_A_TWO_BY_TWO
373 at += 8;
374 }
375 if flags & 0x0020 == 0 { // MORE_COMPONENTS
376 break;
377 }
378 }
379 }
380
381 let mut new_glyf: Vec<u8> = Vec::new();
382 let mut new_loca: Vec<u8> = Vec::with_capacity(4 * (n + 1));
383 for g in 0..n {
384 new_loca.extend_from_slice(&(new_glyf.len() as u32).to_be_bytes());
385 if keep.contains(&(g as u16)) {
386 new_glyf.extend_from_slice(res!(glyph(g)));
387 while new_glyf.len() % 4 != 0 {
388 new_glyf.push(0);
389 }
390 }
391 }
392 new_loca.extend_from_slice(&(new_glyf.len() as u32).to_be_bytes());
393
394 let mut new_head = head.to_vec();
395 if new_head.len() < 54 {
396 return Err(err!("The head table is {} bytes, too short.", new_head.len(); Invalid, Input));
397 }
398 new_head[8..12].copy_from_slice(&[0, 0, 0, 0]); // checkSumAdjustment, recomputed below
399 new_head[50..52].copy_from_slice(&1u16.to_be_bytes()); // long loca
400
401 // The metrics of an emptied glyph are zeroed, so the runs of them compress to almost nothing; a
402 // viewer takes its widths from the PDF's /W, never from here.
403 let mut new_hmtx = match self.table(b"hmtx") {
404 Some(t) => t.to_vec(),
405 None => return Err(err!("The font has no hmtx table."; Invalid, Input, Missing)),
406 };
407 let n_hm = match self.table(b"hhea") {
408 Some(t) => res!(rd_u16(t, 34)) as usize,
409 None => return Err(err!("The font has no hhea table."; Invalid, Input, Missing)),
410 };
411 for g in 0..n {
412 if keep.contains(&(g as u16)) {
413 continue;
414 }
415 let (at, len) = if g < n_hm { (4 * g, 4) } else { (4 * n_hm + 2 * (g - n_hm), 2) };
416 if let Some(span) = new_hmtx.get_mut(at..at + len) {
417 span.fill(0);
418 }
419 }
420
421 let mut out_tables: Vec<([u8; 4], Vec<u8>)> = Vec::new();
422 for tag in [b"cvt ", b"fpgm", b"glyf", b"head", b"hhea", b"hmtx", b"loca", b"maxp", b"prep"] {
423 let body = match tag {
424 b"glyf" => std::mem::take(&mut new_glyf),
425 b"loca" => std::mem::take(&mut new_loca),
426 b"head" => std::mem::take(&mut new_head),
427 b"hmtx" => std::mem::take(&mut new_hmtx),
428 _ => match self.table(tag) {
429 Some(t) => t.to_vec(),
430 None => continue,
431 },
432 };
433 out_tables.push((*tag, body));
434 }
435 Ok(write_sfnt(0x0001_0000, out_tables))
436 }
437}
438
439/// The sum of a table's big-endian words, zero padded, as `sfnt` checksums are taken.
440fn checksum(b: &[u8]) -> u32 {
441 let mut sum = 0u32;
442 for chunk in b.chunks(4) {
443 let mut w = [0u8; 4];
444 w[..chunk.len()].copy_from_slice(chunk);
445 sum = sum.wrapping_add(u32::from_be_bytes(w));
446 }
447 sum
448}
449
450/// Serialises an `sfnt` from tables already sorted by tag, filling in the directory, the checksums and
451/// `head`'s whole-file adjustment.
452fn write_sfnt(version: u32, tables: Vec<([u8; 4], Vec<u8>)>) -> Vec<u8> {
453 let n = tables.len();
454 let mut pow = 1usize;
455 let mut sel = 0u16;
456 while pow * 2 <= n {
457 pow *= 2;
458 sel += 1;
459 }
460 let range = (pow * 16) as u16;
461 let mut out = Vec::new();
462 out.extend_from_slice(&version.to_be_bytes());
463 out.extend_from_slice(&(n as u16).to_be_bytes());
464 out.extend_from_slice(&range.to_be_bytes());
465 out.extend_from_slice(&sel.to_be_bytes());
466 out.extend_from_slice(&((n * 16) as u16).wrapping_sub(range).to_be_bytes());
467
468 let mut off = 12 + 16 * n;
469 let mut head_at = None;
470 for (tag, body) in &tables {
471 out.extend_from_slice(tag);
472 out.extend_from_slice(&checksum(body).to_be_bytes());
473 out.extend_from_slice(&(off as u32).to_be_bytes());
474 out.extend_from_slice(&(body.len() as u32).to_be_bytes());
475 if tag == b"head" {
476 head_at = Some(off);
477 }
478 off += (body.len() + 3) & !3;
479 }
480 for (_, body) in &tables {
481 out.extend_from_slice(body);
482 while out.len() % 4 != 0 {
483 out.push(0);
484 }
485 }
486 if let Some(h) = head_at {
487 let adj = 0xb1b0_afbau32.wrapping_sub(checksum(&out));
488 if let Some(slot) = out.get_mut(h + 8..h + 12) {
489 slot.copy_from_slice(&adj.to_be_bytes());
490 }
491 }
492 out
493}
494
495/// The PostScript name (name id 6), preferring the Windows Unicode record, reduced to the characters
496/// a PDF name may carry unescaped.
497fn postscript_name(d: &[u8], t: Table) -> Option<String> {
498 let count = rd_u16(d, t.off + 2).ok()? as usize;
499 let store = t.off + rd_u16(d, t.off + 4).ok()? as usize;
500 let mut best: Option<String> = None;
501 for i in 0..count {
502 let rec = t.off + 6 + 12 * i;
503 let platform = rd_u16(d, rec).ok()?;
504 let name_id = rd_u16(d, rec + 6).ok()?;
505 let len = rd_u16(d, rec + 8).ok()? as usize;
506 let off = rd_u16(d, rec + 10).ok()? as usize;
507 if name_id != 6 {
508 continue;
509 }
510 let raw = slice(d, store + off, len).ok()?;
511 let s: String = match platform {
512 0 | 3 => {
513 let units: Vec<u16> = raw.chunks(2)
514 .filter(|c| c.len() == 2)
515 .map(|c| ((c[0] as u16) << 8) | c[1] as u16)
516 .collect();
517 String::from_utf16_lossy(&units)
518 },
519 _ => raw.iter().map(|&b| b as char).collect(),
520 };
521 let clean: String = s.chars()
522 .filter(|c| c.is_ascii_graphic() && !"()<>[]{}/%#".contains(*c))
523 .take(63)
524 .collect();
525 if !clean.is_empty() {
526 if platform == 3 {
527 return Some(clean);
528 }
529 if best.is_none() {
530 best = Some(clean);
531 }
532 }
533 }
534 best
535}
536
537// ┌───────────────────────────────────────────────────────────────────────────┐
538// │ CFF │
539// └───────────────────────────────────────────────────────────────────────────┘
540
541// DICT operators this rewriter touches; a two-byte operator `12 x` is `1200 + x`.
542const OP_CHARSET: u16 = 15;
543const OP_ENCODING: u16 = 16;
544const OP_CHARSTRINGS: u16 = 17;
545const OP_PRIVATE: u16 = 18;
546const OP_SUBRS: u16 = 19;
547const OP_ROS: u16 = 1230;
548const OP_CIDCOUNT: u16 = 1234;
549const OP_FDARRAY: u16 = 1236;
550const OP_FDSELECT: u16 = 1237;
551// The dict operators whose operand is a string id: version, Notice, FullName, FamilyName, Weight,
552// Copyright, PostScript, BaseFontName and FontName.
553const SID_OPS: [u16; 9] = [0, 1, 2, 3, 4, 1200, 1221, 1222, 1238];
554
555/// Reads a CFF INDEX at `at`: its items and the offset just past it.
556fn read_index(d: &[u8], at: usize) -> Outcome<(Vec<&[u8]>, usize)> {
557 let count = res!(rd_u16(d, at)) as usize;
558 if count == 0 {
559 return Ok((Vec::new(), at + 2));
560 }
561 let osz = res!(rd_u8(d, at + 2)) as usize;
562 if osz == 0 || osz > 4 {
563 return Err(err!("A CFF INDEX at {} declares offset size {}.", at, osz; Invalid, Input));
564 }
565 let offs_at = at + 3;
566 let data_at = offs_at + (count + 1) * osz - 1; // offsets are one-based
567 let mut offs = Vec::with_capacity(count + 1);
568 for i in 0..=count {
569 offs.push(res!(rd_uint(d, offs_at + i * osz, osz)));
570 }
571 let mut items = Vec::with_capacity(count);
572 for i in 0..count {
573 if offs[i + 1] < offs[i] {
574 return Err(err!("A CFF INDEX at {} has a decreasing offset at item {}.", at, i;
575 Invalid, Input));
576 }
577 items.push(res!(slice(d, data_at + offs[i], offs[i + 1] - offs[i])));
578 }
579 Ok((items, data_at + offs[count]))
580}
581
582fn write_index<T: AsRef<[u8]>>(items: &[T]) -> Vec<u8> {
583 let mut out = Vec::new();
584 out.extend_from_slice(&(items.len() as u16).to_be_bytes());
585 if items.is_empty() {
586 return out;
587 }
588 let total: usize = items.iter().map(|i| i.as_ref().len()).sum::<usize>() + 1;
589 let osz = if total < 0x100 { 1 } else if total < 0x1_0000 { 2 } else if total < 0x100_0000 { 3 } else { 4 };
590 out.push(osz as u8);
591 let mut off = 1usize;
592 let put = |out: &mut Vec<u8>, v: usize| {
593 for k in (0..osz).rev() {
594 out.push((v >> (8 * k)) as u8);
595 }
596 };
597 put(&mut out, off);
598 for it in items {
599 off += it.as_ref().len();
600 put(&mut out, off);
601 }
602 for it in items {
603 out.extend_from_slice(it.as_ref());
604 }
605 out
606}
607
608/// One DICT entry: its operator, its operands' raw bytes (re-emitted untouched when the entry is kept)
609/// and their integer values (the only kind this rewriter needs to read).
610struct DictEntry {
611 op: u16,
612 raw: Vec<u8>,
613 ints: Vec<i64>,
614}
615
616fn parse_dict(d: &[u8]) -> Outcome<Vec<DictEntry>> {
617 let mut out = Vec::new();
618 let mut raw = Vec::new();
619 let mut ints = Vec::new();
620 let mut i = 0;
621 while i < d.len() {
622 let b = d[i];
623 let start = i;
624 match b {
625 0..=21 => {
626 let op = if b == 12 {
627 i += 1;
628 1200 + res!(rd_u8(d, i)) as u16
629 } else {
630 b as u16
631 };
632 i += 1;
633 out.push(DictEntry { op, raw: std::mem::take(&mut raw), ints: std::mem::take(&mut ints) });
634 continue;
635 },
636 28 => {
637 ints.push(res!(rd_i16(d, i + 1)) as i64);
638 i += 3;
639 },
640 29 => {
641 ints.push(res!(rd_u32(d, i + 1)) as i32 as i64);
642 i += 5;
643 },
644 30 => {
645 // A real: nibbles to the terminator 0xf. Its value is never an offset, so it reads as 0.
646 i += 1;
647 loop {
648 let n = res!(rd_u8(d, i));
649 i += 1;
650 if n & 0x0f == 0x0f || n >> 4 == 0x0f {
651 break;
652 }
653 }
654 ints.push(0);
655 },
656 32..=246 => {
657 ints.push(b as i64 - 139);
658 i += 1;
659 },
660 247..=250 => {
661 ints.push((b as i64 - 247) * 256 + res!(rd_u8(d, i + 1)) as i64 + 108);
662 i += 2;
663 },
664 251..=254 => {
665 ints.push(-(b as i64 - 251) * 256 - res!(rd_u8(d, i + 1)) as i64 - 108);
666 i += 2;
667 },
668 _ => return Err(err!("A CFF DICT holds the reserved byte {} at {}.", b, i; Invalid, Input)),
669 }
670 raw.extend_from_slice(res!(slice(d, start, i - start)));
671 }
672 Ok(out)
673}
674
675fn int5(v: usize) -> [u8; 5] {
676 let b = (v as u32).to_be_bytes();
677 [29, b[0], b[1], b[2], b[3]]
678}
679
680fn put_op(out: &mut Vec<u8>, op: u16) {
681 if op >= 1200 {
682 out.push(12);
683 out.push((op - 1200) as u8);
684 } else {
685 out.push(op as u8);
686 }
687}
688
689fn entry_int(entries: &[DictEntry], op: u16, k: usize) -> Option<i64> {
690 entries.iter().find(|e| e.op == op).and_then(|e| e.ints.get(k).copied())
691}
692
693/// The bias a charstring adds to a subroutine number, by the size of the INDEX it calls into.
694fn bias(n: usize) -> i64 {
695 if n < 1240 { 107 } else if n < 33900 { 1131 } else { 32768 }
696}
697
698/// Flattens a Type 2 charstring: every subroutine call is replaced by the body it calls, so the subset
699/// needs no subroutines at all. For the few dozen glyphs a document shows this is far smaller than the
700/// thousands of shared subroutines a whole font carries. The operand stack is followed only as far as
701/// the flattening needs: its depth, for the stem count a `hintmask` sizes its mask by, and the byte where
702/// the last operand began, so a call's subroutine number can be cut back out.
703struct Flatten<'a> {
704 gsubrs: &'a [&'a [u8]],
705 lsubrs: &'a [&'a [u8]],
706 out: Vec<u8>,
707 stems: usize,
708 stack: usize, // operand depth
709 last: i64, // the top operand, the subroutine number a call pops
710 last_at: usize, // where in `out` the top operand's bytes begin
711}
712
713enum Flow {
714 Return,
715 End,
716}
717
718impl<'a> Flatten<'a> {
719
720 fn new(gsubrs: &'a [&'a [u8]], lsubrs: &'a [&'a [u8]]) -> Self {
721 Self { gsubrs, lsubrs, out: Vec::new(), stems: 0, stack: 0, last: 0, last_at: 0 }
722 }
723
724 fn operand(&mut self, cs: &[u8], i: usize, len: usize, v: i64) -> Outcome<()> {
725 self.last = v;
726 self.last_at = self.out.len();
727 self.stack += 1;
728 self.out.extend_from_slice(res!(slice(cs, i, len)));
729 Ok(())
730 }
731
732 fn run(&mut self, cs: &[u8], depth: usize) -> Outcome<Flow> {
733 if depth > 10 {
734 return Err(err!("Charstring subroutines nest past the limit of ten."; Invalid, Input));
735 }
736 let mut i = 0;
737 while i < cs.len() {
738 let b = cs[i];
739 match b {
740 28 => {
741 let v = res!(rd_i16(cs, i + 1)) as i64;
742 res!(self.operand(cs, i, 3, v));
743 i += 3;
744 },
745 32..=246 => {
746 res!(self.operand(cs, i, 1, b as i64 - 139));
747 i += 1;
748 },
749 247..=250 => {
750 let v = (b as i64 - 247) * 256 + res!(rd_u8(cs, i + 1)) as i64 + 108;
751 res!(self.operand(cs, i, 2, v));
752 i += 2;
753 },
754 251..=254 => {
755 let v = -(b as i64 - 251) * 256 - res!(rd_u8(cs, i + 1)) as i64 - 108;
756 res!(self.operand(cs, i, 2, v));
757 i += 2;
758 },
759 255 => {
760 let v = (res!(rd_u32(cs, i + 1)) as i32 >> 16) as i64;
761 res!(self.operand(cs, i, 5, v));
762 i += 5;
763 },
764 1 | 3 | 18 | 23 => { // hstem, vstem, hstemhm, vstemhm
765 self.stems += self.stack / 2;
766 self.stack = 0;
767 self.out.push(b);
768 i += 1;
769 },
770 19 | 20 => { // hintmask, cntrmask: an implicit vstem, then the mask bytes
771 self.stems += self.stack / 2;
772 self.stack = 0;
773 let n = 1 + (self.stems + 7) / 8;
774 self.out.extend_from_slice(res!(slice(cs, i, n)));
775 i += n;
776 },
777 10 | 29 => { // callsubr, callgsubr
778 if self.stack == 0 {
779 return Err(err!("A subroutine call has no operand."; Invalid, Input));
780 }
781 self.stack -= 1;
782 self.out.truncate(self.last_at);
783 let subrs = if b == 29 { self.gsubrs } else { self.lsubrs };
784 let idx = self.last + bias(subrs.len());
785 let body = match usize::try_from(idx).ok().and_then(|k| subrs.get(k)) {
786 Some(s) => *s,
787 None => return Err(err!(
788 "A charstring calls subroutine {} of {}.", idx, subrs.len(); Invalid, Input)),
789 };
790 i += 1;
791 if let Flow::End = res!(self.run(body, depth + 1)) {
792 return Ok(Flow::End);
793 }
794 },
795 11 => return Ok(Flow::Return),
796 14 => {
797 // Four operands (five with a width) is the deprecated `seac`, which names its accent and
798 // base by standard-encoding code -- a name-keyed idea with no meaning in a CID font.
799 if self.stack >= 4 {
800 return Err(err!("The glyph is an accented seac composite, which a CID-keyed \
801 subset cannot express."; Invalid, Input, Unimplemented));
802 }
803 self.out.push(14);
804 return Ok(Flow::End);
805 },
806 12 => {
807 self.stack = 0;
808 self.out.extend_from_slice(res!(slice(cs, i, 2)));
809 i += 2;
810 },
811 _ => {
812 self.stack = 0;
813 self.out.push(b);
814 i += 1;
815 },
816 }
817 }
818 Ok(Flow::Return)
819 }
820}
821
822/// Cuts a CFF table down to `keep` as a CID-keyed font with no subroutines. The kept glyphs are packed
823/// into consecutive glyph slots and the charset maps each slot back to its original glyph id as its CID,
824/// so a PDF still shows a glyph by the id the shaper gave it.
825fn subset_cff(d: &[u8], keep: &BTreeSet<u16>, num_glyphs: usize) -> Outcome<Vec<u8>> {
826 let hdr_size = res!(rd_u8(d, 2)) as usize;
827 let (names, at) = res!(read_index(d, hdr_size));
828 let (tops, at) = res!(read_index(d, at));
829 let (_strings, at) = res!(read_index(d, at));
830 let (gsubrs, _) = res!(read_index(d, at));
831 let top_raw = match tops.first() {
832 Some(t) => *t,
833 None => return Err(err!("The CFF table holds no Top DICT."; Invalid, Input, Missing)),
834 };
835 let top = res!(parse_dict(top_raw));
836
837 let cs_off = match entry_int(&top, OP_CHARSTRINGS, 0) {
838 Some(o) if o > 0 => o as usize,
839 _ => return Err(err!("The CFF Top DICT names no CharStrings."; Invalid, Input)),
840 };
841 let (charstrings, _) = res!(read_index(d, cs_off));
842 if charstrings.len() != num_glyphs {
843 return Err(err!("The CFF holds {} charstrings but maxp counts {} glyphs.",
844 charstrings.len(), num_glyphs; Invalid, Input));
845 }
846
847 // The font dicts, each with its private dict and local subroutines, and which glyph uses which.
848 struct Fd<'a> {
849 dict: Vec<DictEntry>, // the font dict's own entries, Private excluded
850 private: Vec<DictEntry>, // Subrs excluded: the flattened glyphs call none
851 subrs: Vec<&'a [u8]>,
852 }
853 let read_private = |entries: &[DictEntry]| -> Outcome<(Vec<DictEntry>, Vec<&[u8]>)> {
854 let size = entry_int(entries, OP_PRIVATE, 0).unwrap_or(0).max(0) as usize;
855 let off = entry_int(entries, OP_PRIVATE, 1).unwrap_or(0).max(0) as usize;
856 if size == 0 {
857 return Ok((Vec::new(), Vec::new()));
858 }
859 let pd = res!(parse_dict(res!(slice(d, off, size))));
860 let subrs = match entry_int(&pd, OP_SUBRS, 0) {
861 Some(rel) if rel > 0 => res!(read_index(d, off + rel as usize)).0,
862 _ => Vec::new(),
863 };
864 Ok((pd.into_iter().filter(|e| e.op != OP_SUBRS).collect(), subrs))
865 };
866
867 let mut fds: Vec<Fd> = Vec::new();
868 let mut fd_of: Vec<u8> = vec![0; num_glyphs];
869 if top.iter().any(|e| e.op == OP_ROS) {
870 let fda_off = match entry_int(&top, OP_FDARRAY, 0) {
871 Some(o) if o > 0 => o as usize,
872 _ => return Err(err!("A CID-keyed CFF names no FDArray."; Invalid, Input)),
873 };
874 for raw in res!(read_index(d, fda_off)).0 {
875 let fd = res!(parse_dict(raw));
876 let (private, subrs) = res!(read_private(&fd));
877 fds.push(Fd { dict: fd.into_iter().filter(|e| e.op != OP_PRIVATE).collect(), private, subrs });
878 }
879 let sel = match entry_int(&top, OP_FDSELECT, 0) {
880 Some(o) if o > 0 => o as usize,
881 _ => return Err(err!("A CID-keyed CFF names no FDSelect."; Invalid, Input)),
882 };
883 match res!(rd_u8(d, sel)) {
884 0 => for g in 0..num_glyphs {
885 fd_of[g] = res!(rd_u8(d, sel + 1 + g));
886 },
887 3 => {
888 let n = res!(rd_u16(d, sel + 1)) as usize;
889 for r in 0..n {
890 let first = res!(rd_u16(d, sel + 3 + 3 * r)) as usize;
891 let fd = res!(rd_u8(d, sel + 5 + 3 * r));
892 let end = res!(rd_u16(d, sel + 6 + 3 * r)) as usize;
893 for g in first..end.min(num_glyphs) {
894 fd_of[g] = fd;
895 }
896 }
897 },
898 f => return Err(err!("FDSelect format {} is not one CFF defines.", f; Invalid, Input)),
899 }
900 } else {
901 let (private, subrs) = res!(read_private(&top));
902 fds.push(Fd { dict: Vec::new(), private, subrs });
903 }
904
905 // The kept glyphs, flattened, in glyph-id order: slot k holds CID keep[k], and slot 0 is glyph 0.
906 let cids: Vec<u16> = keep.iter().copied().collect();
907 let mut new_cs: Vec<Vec<u8>> = Vec::with_capacity(cids.len());
908 let mut slot_fd: Vec<u8> = Vec::with_capacity(cids.len());
909 for &g in &cids {
910 let fd = fd_of.get(g as usize).copied().unwrap_or(0);
911 let lsubrs: &[&[u8]] = match fds.get(fd as usize) {
912 Some(f) => &f.subrs,
913 None => return Err(err!("Glyph {} names font dict {} of {}.", g, fd, fds.len(); Invalid, Input)),
914 };
915 let mut flat = Flatten::new(&gsubrs, lsubrs);
916 if let Err(e) = flat.run(charstrings[g as usize], 0) {
917 return Err(err!(e, "Glyph {} could not be flattened.", g; Invalid, Input));
918 }
919 new_cs.push(flat.out);
920 slot_fd.push(fd);
921 }
922
923 // Strings: only the registry and ordering of the identity ROS. The originals are mostly glyph names,
924 // which a CID font has no use for; the few a dict names -- a notice, a family name -- are dropped
925 // with the entries that name them.
926 let new_strings: Vec<&[u8]> = vec![b"Adobe", b"Identity"];
927 let sid_adobe = 391;
928 let sid_identity = 392;
929
930 // Charset format 0: each slot past the first names its CID.
931 let mut charset = vec![0u8];
932 for &cid in cids.iter().skip(1) {
933 charset.extend_from_slice(&cid.to_be_bytes());
934 }
935 // FDSelect format 3: one range per run of slots sharing a font dict.
936 let mut ranges: Vec<(u16, u8)> = Vec::new();
937 for (k, &fd) in slot_fd.iter().enumerate() {
938 if ranges.last().map_or(true, |r| r.1 != fd) {
939 ranges.push((k as u16, fd));
940 }
941 }
942 let mut fdselect = vec![3u8];
943 fdselect.extend_from_slice(&(ranges.len() as u16).to_be_bytes());
944 for (first, fd) in &ranges {
945 fdselect.extend_from_slice(&first.to_be_bytes());
946 fdselect.push(*fd);
947 }
948 fdselect.extend_from_slice(&(cids.len() as u16).to_be_bytes());
949
950 let privates: Vec<Vec<u8>> = fds.iter().map(|fd| {
951 let mut p = Vec::new();
952 for e in &fd.private {
953 p.extend_from_slice(&e.raw);
954 put_op(&mut p, e.op);
955 }
956 p
957 }).collect();
958
959 // Two passes: sizes with placeholder offsets, then the same layout with the real ones. Every offset
960 // is a five-byte integer, so the sizes cannot change between the passes.
961 let build_top = |charset_at: usize, cs_at: usize, fda_at: usize, sel_at: usize| -> Vec<u8> {
962 let mut t = Vec::new();
963 t.extend_from_slice(&int5(sid_adobe));
964 t.extend_from_slice(&int5(sid_identity));
965 t.push(139); // supplement 0
966 put_op(&mut t, OP_ROS);
967 for e in &top {
968 if matches!(e.op, OP_CHARSET | OP_ENCODING | OP_CHARSTRINGS | OP_PRIVATE | OP_ROS
969 | OP_CIDCOUNT | OP_FDARRAY | OP_FDSELECT) || SID_OPS.contains(&e.op)
970 {
971 continue;
972 }
973 t.extend_from_slice(&e.raw);
974 put_op(&mut t, e.op);
975 }
976 t.extend_from_slice(&int5(num_glyphs));
977 put_op(&mut t, OP_CIDCOUNT);
978 t.extend_from_slice(&int5(charset_at));
979 put_op(&mut t, OP_CHARSET);
980 t.extend_from_slice(&int5(cs_at));
981 put_op(&mut t, OP_CHARSTRINGS);
982 t.extend_from_slice(&int5(fda_at));
983 put_op(&mut t, OP_FDARRAY);
984 t.extend_from_slice(&int5(sel_at));
985 put_op(&mut t, OP_FDSELECT);
986 t
987 };
988 let build_fda = |priv_at: &[usize]| -> Vec<u8> {
989 let dicts: Vec<Vec<u8>> = fds.iter().enumerate().map(|(k, fd)| {
990 let mut f = Vec::new();
991 for e in fd.dict.iter().filter(|e| !SID_OPS.contains(&e.op)) {
992 f.extend_from_slice(&e.raw);
993 put_op(&mut f, e.op);
994 }
995 f.extend_from_slice(&int5(privates.get(k).map_or(0, |p| p.len())));
996 f.extend_from_slice(&int5(priv_at.get(k).copied().unwrap_or(0)));
997 put_op(&mut f, OP_PRIVATE);
998 f
999 }).collect();
1000 write_index(&dicts)
1001 };
1002
1003 let no_subrs: [&[u8]; 0] = [];
1004 let name_idx = write_index(&names);
1005 let string_idx = write_index(&new_strings);
1006 let gsubr_idx = write_index(&no_subrs);
1007 let cs_idx = write_index(&new_cs);
1008 let top_len = write_index(&[build_top(0, 0, 0, 0)]).len();
1009 let fda_len = build_fda(&vec![0; fds.len()]).len();
1010
1011 let charset_at = 4 + name_idx.len() + top_len + string_idx.len() + gsubr_idx.len();
1012 let sel_at = charset_at + charset.len();
1013 let cs_at = sel_at + fdselect.len();
1014 let fda_at = cs_at + cs_idx.len();
1015 let mut priv_at = Vec::with_capacity(privates.len());
1016 let mut next = fda_at + fda_len;
1017 for p in &privates {
1018 priv_at.push(next);
1019 next += p.len();
1020 }
1021
1022 let mut out = Vec::with_capacity(next);
1023 out.extend_from_slice(&[1, 0, 4, 4]);
1024 out.extend_from_slice(&name_idx);
1025 out.extend_from_slice(&write_index(&[build_top(charset_at, cs_at, fda_at, sel_at)]));
1026 out.extend_from_slice(&string_idx);
1027 out.extend_from_slice(&gsubr_idx);
1028 out.extend_from_slice(&charset);
1029 out.extend_from_slice(&fdselect);
1030 out.extend_from_slice(&cs_idx);
1031 out.extend_from_slice(&build_fda(&priv_at));
1032 for p in &privates {
1033 out.extend_from_slice(p);
1034 }
1035 if out.len() != next {
1036 return Err(err!("The rewritten CFF is {} bytes where its layout planned {}.", out.len(), next; Bug));
1037 }
1038 Ok(out)
1039}
1040
1041#[cfg(test)]
1042mod tests {
1043 use super::*;
1044
1045 #[test]
1046 fn test_an_index_round_trips_00() -> Outcome<()> {
1047 let items: Vec<&[u8]> = vec![b"abc", b"", b"defg"];
1048 let bytes = write_index(&items);
1049 let (back, end) = res!(read_index(&bytes, 0));
1050 assert_eq!(back, items);
1051 assert_eq!(end, bytes.len());
1052 Ok(())
1053 }
1054
1055 #[test]
1056 fn test_a_dict_keeps_its_operands_01() -> Outcome<()> {
1057 // 100 200 Private, then 12 30 (ROS) with three small ints.
1058 let mut d = Vec::new();
1059 d.extend_from_slice(&[239, 247, 92, 18]); // 100, 200, Private
1060 d.extend_from_slice(&[140, 141, 139, 12, 30]);
1061 let e = res!(parse_dict(&d));
1062 assert_eq!(e.len(), 2);
1063 assert_eq!(e[0].op, OP_PRIVATE);
1064 assert_eq!(e[0].ints, vec![100, 200]);
1065 assert_eq!(e[1].op, OP_ROS);
1066 assert_eq!(e[1].ints, vec![1, 2, 0]);
1067 Ok(())
1068 }
1069}