//! A Word document, rendered by this app rather than by somebody else's server.
//!
//! The third option beside Microsoft's and Google's viewers. Those two work by
//! sending the document to a third party and embedding what comes back; this one
//! parses the file in the browser and renders it as ordinary elements.
//!
//! It buys three things the embedded viewers cannot: the document reaches
//! nobody, the text is selectable and findable with the browser's own search,
//! and it reflows on a phone instead of being a fixed page in a scrolling box.
//!
//! It costs fidelity. This is not a pagination engine: there are no page breaks,
//! no margins, no line-breaking to Word's rules. A heading is an `
`, a
//! paragraph is a `
`, a table is a `
`. For minutes and agendas — which
//! is what this wiki holds — that reads better than the real thing. For a
//! carefully laid-out document it will not, and the other two options are still
//! there.
//!
//! Parsing is `docx-parser` (MIT, Yuki Yokotani, github.com/yukiyokotani/
//! office-open-xml-viewer), which hands back the document model as JSON. Only
//! the parts of that model this renders are deserialised; the model carries far
//! more (typography acquisition, font slots, tab stops) than a DOM rendering can
//! use.
use dioxus::prelude::*;
use serde::Deserialize;
use wasm_bindgen::JsCast;
/// One block in the document body.
///
/// The parser tags these with `type`, and the two that matter are paragraphs and
/// tables. Anything else is skipped rather than guessed at — a block this does
/// not understand is better absent than rendered wrong.
// A paragraph is much the larger of the two, and boxing it would put a pointer
// chase in front of every run of every line for a document that is deserialised
// once and read from thereafter.
#[allow(clippy::large_enum_variant)]
#[derive(Deserialize, Clone, Debug, PartialEq)]
#[serde(tag = "type")]
pub enum Block {
#[serde(rename = "paragraph")]
Paragraph(Paragraph),
#[serde(rename = "table")]
Table(Table),
#[serde(other)]
Unknown,
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Paragraph {
#[serde(default)]
pub runs: Vec,
/// `Heading1`, `Title`, … when the document uses styles.
#[serde(default)]
pub style_id: Option,
/// 0-8 for headings 1-9 in OOXML; 9 (or absent) is body text.
#[serde(default)]
pub outline_level: Option,
#[serde(default)]
pub alignment: Option,
/// Boxed because it is by far the largest thing a paragraph carries, and a
/// paragraph is one variant of [`Block`]: unboxed it made that enum as big
/// as its biggest member for every block in a document, tables included.
#[serde(default)]
pub numbering: Option>,
#[serde(default)]
pub indent_left: Option,
/// The FIRST line's extra indent, separate from `indent_left` and usually
/// negative: a hanging indent. See [`paragraph_style`].
#[serde(default)]
pub indent_first: Option,
/// Space above and below the paragraph, in points, as the document asks for
/// it. Often zero: a document that separates its paragraphs with blank ones
/// wants no space at all, and adding some doubles every gap it has.
#[serde(default)]
pub space_before: Option,
#[serde(default)]
pub space_after: Option,
#[serde(default)]
pub line_spacing: Option,
/// The size this paragraph's text ends up at, in points, resolved by the
/// parser through the style chain AND the document defaults. Read for the
/// pagination, which needs the size Word would actually set: a document can
/// state a size on almost no run at all -- one here states one on five runs
/// of a hundred and eighty, and those five are its headings -- so taking
/// the commonest size the RUNS state measures the whole document at its
/// heading size.
#[serde(default)]
pub default_font_size: Option,
/// The typeface Word resolves for this paragraph, through the style chain
/// and the theme. Read for the measuring, which has to lay the text out in
/// the face the document sets: one face for every document was tried, and a
/// Times New Roman document measured in Calibri's metrics came out a tenth
/// too tall, which is a whole page.
#[serde(default)]
pub default_font_family: Option,
/// Word suppresses the space between adjacent paragraphs of the same style
/// when this is set, which is how a bulleted list closes up. Resolved
/// through the style chain. Without it a list measures 8pt per item taller
/// than Word lays it out.
#[serde(default)]
pub contextual_spacing: bool,
/// Word keeps this paragraph on the same page as the one after it, which
/// every heading style does. A heading is exactly what a page tends to
/// break after, and Word moves it down with the text it introduces rather
/// than leaving it stranded at the foot of a page.
#[serde(default)]
pub keep_next: bool,
/// How big this heading is against the document's own body text, as a
/// multiplier. Not from the document: computed by [`scale_headings`] after
/// parsing, because it takes the whole document to know what "body text"
/// means here.
#[serde(skip)]
pub heading_scale: Option,
}
/// A paragraph's line spacing.
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct LineSpacing {
/// A multiplier when `rule` is `auto`; a measurement in points otherwise.
#[serde(default)]
pub value: f64,
/// `auto`, `exact` or `atLeast`.
#[serde(default)]
pub rule: String,
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Numbering {
/// `bullet` or a number format (`decimal`, `lowerLetter`, …).
#[serde(default)]
pub format: Option,
#[serde(default)]
pub level: Option,
/// A list whose bullet is a picture rather than a character. Word calls it
/// `numPicBullet`; the browser's own disc is not it.
#[serde(default)]
pub pic_bullet_image_path: Option,
#[serde(default)]
pub pic_bullet_mime_type: Option,
#[serde(default)]
pub pic_bullet_width_pt: Option,
#[serde(default)]
pub pic_bullet_height_pt: Option,
/// Filled in by [`attach_images`], like [`Run::src`].
#[serde(skip)]
pub src: Option>,
}
impl Numbering {
/// The picture this list uses for its bullet, and its format.
pub fn picture(&self) -> Option<(&str, &str)> {
let path = self
.pic_bullet_image_path
.as_deref()
.filter(|p| !p.is_empty())?;
Some((path, self.pic_bullet_mime_type.as_deref().unwrap_or("")))
}
/// The SHAPE of that bullet. Not its size.
///
/// The numbering definition carries a size, and it is a page-layout number
/// from whatever template the document came out of: the one in this wiki
/// asks for an 18pt bullet beside 11pt text, twice the size of the very
/// same image used as a bullet elsewhere in the same document. Drawing that
/// literally is what Word does and what a paginating renderer should do.
/// This renderer is not one, and says so: it favours a document that reads
/// over a document that is reproduced.
///
/// So the stylesheet sets the height, in `em`, and a bullet tracks the text
/// it leads and the reader's own font size. All that is needed from the
/// document is the aspect ratio, so the picture is not squashed.
pub fn bullet_style(&self) -> String {
match (
self.pic_bullet_width_pt.filter(|w| *w > 0.0),
self.pic_bullet_height_pt.filter(|h| *h > 0.0),
) {
(Some(w), Some(h)) => format!("aspect-ratio:{w:.2}/{h:.2};"),
_ => String::new(),
}
}
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Run {
/// What the parser tagged this run as. Only `break` matters here: OOXML
/// writes a line break as a run of its OWN (``), carrying no text, so
/// without reading this it arrived as an empty run and vanished. That is
/// what dropped the newline after a bold lead-in line.
#[serde(default, rename = "type")]
pub kind: Option,
#[serde(default)]
pub text: String,
#[serde(default)]
pub bold: bool,
#[serde(default)]
pub italic: bool,
/// Kept as a loose value because the shape varies: this parser emits a
/// bool, OOXML itself carries a style name (`single`, `dotted`, `none`), and
/// some producers write an object. [`is_underlined`] is what asks.
#[serde(default)]
pub underline: Option,
#[serde(default)]
pub strikethrough: bool,
/// `RRGGBB`, or `auto` — which means "whatever the reader's theme says", so
/// it must NOT be turned into a colour.
#[serde(default)]
pub color: Option,
#[serde(default)]
pub vert_align: Option,
#[serde(default)]
pub hyperlink: Option,
/// Points, as the document resolved them through its styles.
#[serde(default)]
pub font_size: Option,
// --- pictures ---
/// Zip path of the raster the run draws, `word/media/image1.png`. A run
/// either has this or has text; the parser writes one node per picture.
#[serde(default)]
pub image_path: Option,
/// The vector original, when Word kept one beside the raster fallback.
/// Preferred: it scales to the reader's screen instead of blurring.
#[serde(default)]
pub svg_image_path: Option,
#[serde(default)]
pub mime_type: Option,
/// The size Word gives the picture, in points. Points are a CSS unit too,
/// and the same one, so these carry across untouched.
#[serde(default)]
pub width_pt: Option,
#[serde(default)]
pub height_pt: Option,
#[serde(default)]
pub rotation: Option,
#[serde(default)]
pub flip_h: bool,
#[serde(default)]
pub flip_v: bool,
/// The picture itself, as a `data:` url, filled in by [`attach_images`]
/// after parsing. Not from the document: the model carries a path into the
/// zip, and the bytes have to be fetched out of it separately.
///
/// `Rc` because one picture is often drawn many times — a bullet glyph, a
/// logo in a header — and a megabyte of base64 should be stored once
/// however many runs point at it.
#[serde(skip)]
pub src: Option>,
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Table {
#[serde(default)]
pub rows: Vec,
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Row {
#[serde(default)]
pub cells: Vec,
#[serde(default)]
pub is_header: bool,
/// `w:cantSplit`: the document forbids Word to break this row across two
/// pages. Word splits a row by default, and where it splits one the page
/// below it is full rather than ending early.
#[serde(default)]
pub cant_split: bool,
/// `w:trHeight`, in points: a height the document asks this row to have,
/// which is usually the trace of someone dragging its boundary in Word. It
/// is a MINIMUM here whatever the rule says -- a row is never made shorter
/// than its text, and clipping the text is not something a reader that
/// reflows should do.
#[serde(default)]
pub row_height: Option,
}
#[derive(Deserialize, Clone, Debug, PartialEq, Default)]
#[serde(rename_all = "camelCase")]
pub struct Cell {
/// A cell holds blocks, not text: a table cell can contain paragraphs and
/// further tables, so rendering recurses here.
#[serde(default)]
pub content: Vec,
#[serde(default = "one")]
pub col_span: u32,
/// `continue` on a cell that is covered by a vertical merge from above; such
/// a cell is not drawn at all, or the row grows an extra column.
#[serde(default)]
pub v_merge: Option,
/// The cell's shading, as `RRGGBB`. A table's header row is usually the only
/// thing telling a reader it is a header, so this is not decoration.
#[serde(default)]
pub background: Option,
/// The width Word prefers for this column, in points. NOT used to lay the
/// measuring copy out: forcing the columns to these widths was tried and
/// measured, and it made every document worse -- one table's columns come
/// to 670px against a 642px text column, so pinning them squeezes every row
/// taller and an eight-page document measured ten. Kept because the field
/// is what the document says; whatever uses it will have to reconcile that
/// overflow the way Word does.
#[serde(default)]
pub width_pt: Option,
}
/// A cell's shading, as CSS.
///
/// `auto` means "whatever the reader's theme says", exactly as it does on a
/// run's colour, so it must NOT become a colour — a document that says `auto`
/// on a dark theme wants the dark background, not a white one painted over it.
pub fn cell_style(cell: &Cell) -> String {
cell_colour(cell)
}
/// What a cell is painted, if the document paints it.
fn cell_colour(cell: &Cell) -> String {
match cell.background.as_deref() {
Some(hex) if is_real_colour(hex) => {
let hex = hex.trim().trim_start_matches('#');
// Ink to match. A document that paints a header dark navy expects
// light text on it, and the theme's own on-surface colour is not
// that: the background is the DOCUMENT'S, not the theme's, so what
// reads against it has to be worked out from it. A run carrying its
// own colour still wins — this only sets what the cell inherits.
format!("background:#{hex};color:{};", readable_ink(hex))
}
_ => String::new(),
}
}
/// Black or white, whichever can be read on `hex`.
///
/// Relative luminance, as WCAG defines it: the eye is far more sensitive to
/// green than to blue, so a flat average of the channels calls a saturated blue
/// light when it is not.
pub fn readable_ink(hex: &str) -> &'static str {
let channel = |i: usize| u8::from_str_radix(&hex[i..i + 2], 16).unwrap_or(0) as f64 / 255.0;
let linear = |c: f64| match c <= 0.03928 {
true => c / 12.92,
false => ((c + 0.055) / 1.055).powf(2.4),
};
if hex.len() < 6 {
return "#000";
}
let l = 0.2126 * linear(channel(0)) + 0.7152 * linear(channel(2)) + 0.0722 * linear(channel(4));
// 0.179 is where white and black give the same contrast ratio against a
// background, so it is the crossover.
match l > 0.179 {
true => "#000",
false => "#fff",
}
}
/// Whether a colour from the document is an actual colour.
pub fn is_real_colour(hex: &str) -> bool {
let hex = hex.trim().trim_start_matches('#');
!hex.eq_ignore_ascii_case("auto")
&& matches!(hex.len(), 6 | 8)
&& hex.chars().all(|c| c.is_ascii_hexdigit())
}
fn one() -> u32 {
1
}
/// Pictures the document embeds, keyed by their path inside the package.
pub type Images = std::collections::HashMap>;
/// Formats a browser will draw.
///
/// Word embeds whatever it was given, and two of the things it is given
/// regularly — EMF and WMF, Windows' own vector formats — no browser has ever
/// displayed. Those are left for [`render_gaps`](super::render_gaps) to report
/// rather than turned into a broken image icon. Every picture in this wiki's
/// documents is a PNG or a JPEG, but that is a fact about today's documents.
pub fn is_drawable(mime: &str) -> bool {
matches!(
mime.trim().to_ascii_lowercase().as_str(),
"image/png"
| "image/jpeg"
| "image/jpg"
| "image/gif"
| "image/webp"
| "image/avif"
| "image/bmp"
| "image/svg+xml"
)
}
/// Which picture a run draws, and in what format: the vector original when Word
/// kept one, otherwise the raster.
///
/// The SVG sibling has no `mimeType` of its own in the model — the field
/// describes the raster fallback — so it is named here.
pub fn picture_of(run: &Run) -> Option<(&str, &str)> {
if let Some(svg) = run.svg_image_path.as_deref().filter(|p| !p.is_empty()) {
return Some((svg, "image/svg+xml"));
}
let path = run.image_path.as_deref().filter(|p| !p.is_empty())?;
Some((path, run.mime_type.as_deref().unwrap_or("")))
}
/// Read every picture the document draws out of the package, once each.
///
/// Deduplicated on the way in: a bullet glyph is one file referenced from every
/// list item, and a document here draws the same two 1 KB JPEGs fifteen times.
/// Undrawable formats are skipped rather than embedded as bytes no browser can
/// decode; so is anything the package turns out not to contain.
/// The metafile pictures a document draws, for the backend to render.
pub fn collect_metafiles(blocks: &[Block]) -> Vec<(String, String)> {
let mut wanted: Vec<(String, String)> = Vec::new();
let mut want = |picture: Option<(&str, &str)>| {
if let Some((path, mime)) = picture {
if is_metafile(mime) && !wanted.iter().any(|(p, _)| p == path) {
wanted.push((path.to_string(), mime.to_string()));
}
}
};
for_each_paragraph(blocks, &mut |p| {
want(p.numbering.as_ref().and_then(|n| n.picture()));
p.runs.iter().for_each(|run| want(picture_of(run)));
});
wanted
}
pub fn collect_images(blocks: &[Block], package: &[u8]) -> Images {
let mut wanted: Vec<(String, String)> = Vec::new();
let mut want = |picture: Option<(&str, &str)>| {
if let Some((path, mime)) = picture {
if is_drawable(mime) && !wanted.iter().any(|(p, _)| p == path) {
wanted.push((path.to_string(), mime.to_string()));
}
}
};
for_each_paragraph(blocks, &mut |p| {
// A list's bullet is a picture too, and it is named on the paragraph
// rather than in its runs.
want(p.numbering.as_ref().and_then(|n| n.picture()));
p.runs.iter().for_each(|run| want(picture_of(run)));
});
read_images(wanted, package)
}
/// Read a list of already-deduplicated pictures out of a package.
///
/// Shared with the slide renderer, which finds its pictures differently but
/// stores them the same way — a zip path and a mime, out of an OOXML package.
pub fn read_images(wanted: Vec<(String, String)>, package: &[u8]) -> Images {
if wanted.is_empty() {
return Images::new();
}
// One archive for all of them: opening the zip per picture would re-scan the
// central directory each time.
let Ok(mut zip) = zip::ZipArchive::new(std::io::Cursor::new(package)) else {
return Images::new();
};
let mut out = Images::new();
for (path, mime) in wanted {
use std::io::Read;
let mut bytes = Vec::new();
let read = zip
.by_name(&path)
.map(|mut entry| entry.read_to_end(&mut bytes))
.is_ok();
if read && !bytes.is_empty() {
out.insert(path, data_url(&mime, &bytes).into());
}
}
out
}
/// Windows' own vector formats, which Word and PowerPoint keep pasted figures
/// in and which no browser draws.
///
/// Not [`is_drawable`], because nothing here can draw them: they go to the
/// backend, which renders them to PNG (see [`render_metafiles`]).
pub fn is_metafile(mime: &str) -> bool {
matches!(
mime.trim().to_ascii_lowercase().as_str(),
"image/x-emf" | "image/emf" | "image/x-wmf" | "image/wmf" | "application/x-msmetafile"
)
}
/// Every metafile picture a document draws, rendered to PNG by the backend.
///
/// One request each, and a failure is simply left out: the viewer already draws
/// a labelled placeholder in the shape's box for a picture it has no pixels
/// for, which is the right answer for a figure this cannot render either.
pub async fn render_metafiles(
wanted: Vec<(String, String)>,
package: &[u8],
token: Option<&str>,
) -> Images {
let mut out = Images::new();
if wanted.is_empty() {
return out;
}
let Ok(mut zip) = zip::ZipArchive::new(std::io::Cursor::new(package)) else {
return out;
};
for (path, _mime) in wanted {
use std::io::Read;
let mut bytes = Vec::new();
let read = zip
.by_name(&path)
.map(|mut entry| entry.read_to_end(&mut bytes))
.is_ok();
if !read || bytes.is_empty() {
continue;
}
match crate::backend_api::render_metafile(&bytes, token.unwrap_or_default()).await {
Ok((drawn, mime)) => {
out.insert(path, data_url(&mime, &drawn).into());
}
Err(e) => log::info!("metafile {path} not rendered: {e}"),
}
}
out
}
/// A `data:` url for one picture.
fn data_url(mime: &str, bytes: &[u8]) -> String {
use base64::Engine as _;
format!(
"data:{mime};base64,{}",
base64::engine::general_purpose::STANDARD.encode(bytes)
)
}
/// The size of a paragraph's text: whichever size most of its characters are
/// set in. A paragraph is usually all one size, and where it is not, the run
/// carrying the words is the one that matters, not a stray space.
fn dominant_size(p: &Paragraph) -> Option {
let mut weights: Vec<(f64, usize)> = Vec::new();
for run in &p.runs {
let Some(pt) = run.font_size.filter(|s| *s > 0.0) else {
continue;
};
let chars = run.text.trim().chars().count();
if chars == 0 {
continue;
}
match weights.iter_mut().find(|(s, _)| (*s - pt).abs() < 0.01) {
Some((_, n)) => *n += chars,
None => weights.push((pt, chars)),
}
}
weights.into_iter().max_by_key(|(_, n)| *n).map(|(s, _)| s)
}
/// Work out how prominent each heading is, against the document's own body text.
///
/// A Word heading keeps its level and its size separately, and the two do not
/// track each other. The document this was written for has a 20pt `Titel` and
/// an 11pt `Overskrift1` — both of which are outline level 0, so both become an
/// `
`, and rendering them at the `
` size of the app's type scale makes
/// them identical and loses the hierarchy the author wrote. The 11pt one also
/// arrives three times the size it has in the document.
///
/// So a heading is sized RELATIVE to the document's body text rather than
/// absolutely: `20/11` is a big heading and `11/11` is a heading that is bold
/// and no larger, which is exactly what each of those is in Word. Relative
/// rather than absolute so the reader's own font size still sets the scale — an
/// 8pt document should not arrive at 8pt on a phone.
///
/// Body text is the size most of the document's characters are set in, which is
/// what "body text" means. Headings are excluded from that count, or a document
/// of mostly headings would measure itself against its own headings.
///
/// Clamped below at 1: a heading is never SMALLER than the text it introduces,
/// whatever the document says, because the tag has already promised a reader
/// that it is a heading.
pub fn scale_headings(blocks: &mut [Block]) {
let mut sizes: Vec<(f64, usize)> = Vec::new();
for_each_paragraph(blocks, &mut |p| {
if heading_level(p.style_id.as_deref(), p.outline_level).is_some() {
return;
}
let Some(pt) = dominant_size(p) else { return };
let chars = p.runs.iter().map(|r| r.text.trim().chars().count()).sum();
match sizes.iter_mut().find(|(s, _)| (*s - pt).abs() < 0.01) {
Some((_, n)) => *n += chars,
None => sizes.push((pt, chars)),
}
});
let Some(body) = sizes
.into_iter()
.max_by_key(|(_, n)| *n)
.map(|(s, _)| s)
.filter(|s| *s > 0.0)
else {
return;
};
for_each_paragraph_mut(blocks, &mut |p| {
if heading_level(p.style_id.as_deref(), p.outline_level).is_none() {
return;
}
p.heading_scale = dominant_size(p).map(|pt| (pt / body).clamp(1.0, 2.5));
});
}
/// Give every picture run the bytes it draws.
///
/// Separate from parsing because the two come from different places: the model
/// carries a path into the package, and the package is the bytes that were
/// parsed. A run whose picture is missing or undrawable keeps `src: None`, which
/// renders as nothing at all rather than as a broken image.
pub fn attach_images(blocks: &mut [Block], images: &Images) {
for_each_paragraph_mut(blocks, &mut |p| {
if let Some(n) = p.numbering.as_mut() {
if let Some(path) = n.pic_bullet_image_path.clone() {
n.src = images.get(&path).cloned();
}
}
for run in &mut p.runs {
if let Some((path, _)) = picture_of(run) {
run.src = images.get(path).cloned();
}
}
});
}
/// Every paragraph in the tree, table cells included.
fn for_each_paragraph(blocks: &[Block], f: &mut impl FnMut(&Paragraph)) {
for block in blocks {
match block {
Block::Paragraph(p) => f(p),
Block::Table(t) => {
for row in &t.rows {
for cell in &row.cells {
for_each_paragraph(&cell.content, f);
}
}
}
Block::Unknown => {}
}
}
}
fn for_each_paragraph_mut(blocks: &mut [Block], f: &mut impl FnMut(&mut Paragraph)) {
for block in blocks {
match block {
Block::Paragraph(p) => f(p),
Block::Table(t) => {
for row in &mut t.rows {
for cell in &mut row.cells {
for_each_paragraph_mut(&mut cell.content, f);
}
}
}
Block::Unknown => {}
}
}
}
/// How to draw one picture.
///
/// Word's size is in points, which is a CSS unit and the same one, so it
/// carries across exactly. Two things are added:
///
/// * `max-width: 100%` and a matching `aspect-ratio`, so a picture wider than a
/// phone shrinks instead of pushing the document sideways — and shrinks in
/// proportion, which setting both a width and a height in points would not.
/// * the rotation and mirroring Word recorded, as one transform.
pub fn image_style(run: &Run) -> String {
let mut css = String::from("max-width:100%;");
let (w, h) = (
run.width_pt.filter(|w| *w > 0.0),
run.height_pt.filter(|h| *h > 0.0),
);
if let Some(w) = w {
css.push_str(&format!("width:{w:.2}pt;"));
}
match (w, h) {
// The ratio does the work of the height: it survives the shrink.
(Some(w), Some(h)) => css.push_str(&format!("aspect-ratio:{w:.2}/{h:.2};height:auto;")),
(None, Some(h)) => css.push_str(&format!("height:{h:.2}pt;")),
_ => {}
}
let mut transform = String::new();
if let Some(deg) = run.rotation.filter(|d| d.abs() > 0.01) {
transform.push_str(&format!("rotate({deg:.2}deg) "));
}
// Two mirrors are a rotation, and scale(-1,-1) says so; no special case.
if run.flip_h || run.flip_v {
let x = if run.flip_h { -1 } else { 1 };
let y = if run.flip_v { -1 } else { 1 };
transform.push_str(&format!("scale({x},{y})"));
}
if !transform.trim().is_empty() {
css.push_str(&format!("transform:{};", transform.trim()));
}
css
}
/// Which heading a paragraph is, if any.
///
/// Two sources and they disagree often enough to matter: `outlineLevel` is the
/// OOXML numbering (0 = Heading 1), but a document that never sets it still has
/// `styleId` = `Heading2`. Style wins when it parses, because a document that
/// names its styles means them; outline level is the fallback.
///
/// 9 is OOXML's "body text" outline level, and is deliberately not a heading.
///
/// Style names are LOCALISED. Word writes them in the language of the copy that
/// made the document, so a Danish document has `Titel` and `Overskrift1` where
/// an English one has `Title` and `Heading1` — and this wiki is Danish. Matching
/// only the English names left the title of a document rendering as an ordinary
/// paragraph, which is how this was found. Outline level catches the numbered
/// ones in any language; the plain title has no outline level and needs its
/// name read.
pub fn heading_level(style_id: Option<&str>, outline_level: Option) -> Option {
if let Some(style) = style_id {
let s = style.to_ascii_lowercase().replace([' ', '-', '_'], "");
// Trailing digits are the level; what is left is the name.
let digits = s.len() - s.trim_end_matches(|c: char| c.is_ascii_digit()).len();
let (name, level) = s.split_at(s.len() - digits);
let level = level.parse::().ok();
match (KNOWN_HEADING_NAMES.contains(&name), level) {
// `Heading2`, `Overskrift2`, `Titre 2` …
(true, Some(n)) if (1..=9).contains(&n) => return Some(n.min(6)),
// `Title`, `Titel`, `Titre` — a document's one top heading.
(true, None) => return Some(1),
_ => {}
}
}
match outline_level {
Some(l) if (0..=8).contains(&l) => Some(((l + 1) as u8).min(6)),
_ => None,
}
}
/// What Word calls a heading, in the languages a document in this wiki plausibly
/// came from. Lowercased with spaces and dashes removed, and with any trailing
/// level number already stripped.
///
/// Not exhaustive and not meant to be: a heading with a level also carries an
/// outline level, which needs no translating and is the fallback. This list is
/// what rescues the LEVELLESS title, and Danish is the one that matters here.
const KNOWN_HEADING_NAMES: &[&str] = &[
"heading",
"title", // English
"overskrift",
"titel", // Danish, Norwegian, German, Dutch, Swedish
"rubrik", // Swedish
"berschrift", // German, once the umlaut is not an ascii letter
"titre", // French, which uses one word for both
"titulo",
"encabezado", // Spanish, Portuguese
"titolo", // Italian
"otsikko", // Finnish
"naglowek", // Polish
];
/// Whether a run is actually underlined.
///
/// The field was once tested with `.is_some()`, which underlined every run in
/// every document: this parser writes `false` for a run WITHOUT an underline,
/// and `Some(false)` is very much some. A loose type plus a truthiness check is
/// a bug waiting to be reported, and it was.
///
/// Handles the shapes an underline is written in: a bool, a style name where
/// `none` means none, or an object carrying that name under `val`.
pub fn is_underlined(value: Option<&serde_json::Value>) -> bool {
match value {
None | Some(serde_json::Value::Null) => false,
Some(serde_json::Value::Bool(on)) => *on,
Some(serde_json::Value::String(style)) => !matches!(style.as_str(), "" | "none"),
Some(serde_json::Value::Object(map)) => match map.get("val") {
Some(serde_json::Value::String(style)) => !matches!(style.as_str(), "" | "none"),
Some(serde_json::Value::Bool(on)) => *on,
// An object with no `val` at all still says "there is an underline
// here" more than it says there is not.
None => !map.is_empty(),
_ => false,
},
_ => false,
}
}
/// The inline CSS for a run, or an empty string when it needs none.
///
/// Bold and italic are elements rather than styles (see `RunSpan`), so this
/// carries only what has no element of its own.
pub fn run_style(run: &Run) -> String {
let mut css = String::new();
// `auto` is not a colour: it means the consumer decides, and forcing it to
// black would make a document unreadable in the dark theme.
//
// Nor, in practice, is literal black. Word writes 000000 for ordinary body
// text as readily as it writes auto — this handlingsplan states it on 15 of
// its list paragraphs — and a reader in the dark theme got black text on a
// dark surface. Dropping it lets the text inherit the surface it is
// actually on, which is the theme's ink, or a shaded table cell's own
// readable ink where there is one.
//
// The cost is a document that deliberately set black against something
// light this renderer does not paint. It could not have shown that in the
// dark theme anyway, and every other colour is still honoured exactly.
if let Some(c) = run.color.as_deref() {
let default_ink = c == "auto" || c == "000000";
if !default_ink && c.len() == 6 && c.chars().all(|ch| ch.is_ascii_hexdigit()) {
css.push_str(&format!("color:#{c};"));
}
}
match run.vert_align.as_deref() {
Some("superscript") => css.push_str("vertical-align:super;font-size:0.8em;"),
Some("subscript") => css.push_str("vertical-align:sub;font-size:0.8em;"),
_ => {}
}
// Decoration rather than elements, and both at once when both are set. The
// element chain this replaced could only pick one, so a bold underlined run
// silently lost its underline.
match (is_underlined(run.underline.as_ref()), run.strikethrough) {
(true, true) => css.push_str("text-decoration:underline line-through;"),
(true, false) => css.push_str("text-decoration:underline;"),
(false, true) => css.push_str("text-decoration:line-through;"),
(false, false) => {}
}
css
}
/// The inline CSS for a paragraph: alignment and indent, which are the two that
/// change how a document READS rather than merely how it looks.
pub fn paragraph_style(p: &Paragraph) -> String {
let mut css = String::new();
// `left` is emitted, not skipped. It was skipped as "the default", but a
// document does not render in a vacuum: the file viewer around it centres
// its contents (for a centred image, and for the no-preview state), so a
// left-aligned paragraph that says nothing INHERITS centre. Reported from a
// left-aligned document that rendered centred.
match p.alignment.as_deref() {
Some("center") => css.push_str("text-align:center;"),
Some("right") => css.push_str("text-align:right;"),
Some("both") | Some("justify") => css.push_str("text-align:justify;"),
Some("left") | Some("start") => css.push_str("text-align:left;"),
_ => {}
}
// Points to rem against a 16px root, so an indent scales with the reader's
// font size instead of being pinned to Word's. Both indents go through the
// same scale, or the hanging indent below would not line up.
let left = p.indent_left.unwrap_or(0.0).max(0.0);
// A first-line indent, which Word writes as a SEPARATE number and which is
// usually NEGATIVE: that is a hanging indent, and it is how a bulleted
// paragraph is written without being a list. `indentLeft` is then where the
// WRAPPED lines go, and the first line is pulled back to sit under the
// bullet. Rendering only `indentLeft` puts such a paragraph 18pt to the
// right of its neighbours — reported from a document whose bullets were all
// at one level and rendered at two.
//
// CSS says this in one property. Clamped so a first line can never escape
// to the left of the document and be clipped.
let first = p.indent_first.unwrap_or(0.0).max(-left);
if left > 0.0 {
css.push_str(&format!("margin-left:{:.2}rem;", left / 16.0));
}
if first.abs() > 0.01 {
css.push_str(&format!("text-indent:{:.2}rem;", first / 16.0));
}
// Space above and below, when the document says. It usually does — all 44
// paragraphs of the document this was reported from — and what it usually
// says here is ZERO, because it separates its paragraphs with blank ones
// instead. The stylesheet's own comfortable margin is then a SECOND gap on
// top of the blank line, and every space in the document is twice what it
// should be. A document that says nothing keeps the stylesheet's margin.
//
// Headings are left out of this on purpose. A heading is not rendered as
// itself: it becomes an `
`-`
` in the app's own type scale, and its
// rhythm belongs to that scale rather than to points from a Word template.
// The document this came from asks for zero space around its Heading 1,
// which in Word sits under a blank paragraph and here would sit flush
// against the text above it.
let is_heading = heading_level(p.style_id.as_deref(), p.outline_level).is_some();
// A heading takes its prominence from the document, not from the tag: see
// [`scale_headings`]. `em`, so it is measured against the reader's own text
// size rather than against Word's page.
if is_heading {
if let Some(scale) = p.heading_scale.filter(|s| *s > 0.0) {
css.push_str(&format!("font-size:{scale:.3}em;"));
}
}
if !is_heading {
if let Some(pt) = p.space_before.filter(|v| *v >= 0.0) {
css.push_str(&format!("margin-top:{:.2}rem;", pt / 16.0));
}
if let Some(pt) = p.space_after.filter(|v| *v >= 0.0) {
css.push_str(&format!("margin-bottom:{:.2}rem;", pt / 16.0));
}
}
// Line spacing, but only the multiplier form. `exact` and `atLeast` are
// measurements for a fixed page; honouring them in a reflowing column would
// clip a line that wraps differently than Word intended.
if let Some(ls) = p.line_spacing.as_ref() {
if ls.rule == "auto" && ls.value > 0.0 {
css.push_str(&format!("line-height:{:.2};", ls.value));
}
}
css.push_str(&measured_style(p));
css
}
/// What Word resolved for this paragraph, for the off-screen copy that works out
/// where the pages end: the spacing, the size, the face and the indent.
///
/// All of it in custom properties, read only inside `.docx-measure`, so it
/// changes nothing on screen. That is what lets a list item carry its own
/// measurements without moving on the page: a list is indented for READING here,
/// half as deep as Word indents it, and the reader's depth is the right one to
/// read at and the wrong one to paginate by.
pub fn measured_style(p: &Paragraph) -> String {
let mut css = String::new();
// What THIS paragraph leaves under itself and sets its lines at, as Word
// resolved it.
//
// Per paragraph, not per document: a document-wide figure was tried and the
// cells outvoted the body. One file's hundred and thirty cell paragraphs
// leave nothing under themselves and its hundred and eight body paragraphs
// leave eight points, so "what most paragraphs do" was nothing at all, and
// the whole document measured a page short of what Word makes of it.
css.push_str(&format!(
"--p-before:{:.2}pt;--p-after:{:.2}pt;",
p.space_before.filter(|v| *v >= 0.0).unwrap_or(0.0),
p.space_after.filter(|v| *v >= 0.0).unwrap_or(0.0)
));
// And the size Word resolved for this paragraph. Headings especially: this
// renderer sizes them by how prominent they are against the body, which is
// right for reading and wrong for measuring, and a document whose table
// cells are full of headings was measured a quarter short because of it.
if let Some(pt) = p.default_font_size.filter(|v| *v > 0.0) {
css.push_str(&format!("--p-size:{pt:.2}pt;"));
}
// And the face, per paragraph. Not one face for the document: a document
// whose list style is Cambria and whose body is Calibri is most of the
// corpus, and measuring all of it in whichever face is commonest gets the
// other one wrong. The line box comes with it -- what Word calls single
// spacing is a property of the face, and the two shipped faces have theirs
// measured at runtime.
if let Some(named) = p
.default_font_family
.as_deref()
.map(str::trim)
.filter(|f| !f.is_empty())
{
css.push_str(&format!("--p-face:{};", measuring_face(Some(named))));
css.push_str(&format!(
"--p-single:{};",
Measured::of(Some(named)).single()
));
}
let line = match p.line_spacing.as_ref() {
Some(ls) if ls.rule == "auto" && ls.value > 0.0 => ls.value,
// Stating none is stating single.
_ => 1.0,
};
css.push_str(&format!("--p-line:{line:.3};"));
// Where Word's own margin puts this paragraph, in points, which is not what
// the visible margin above says: that one is in rem so an indent scales with
// the reader's text, and at a 16px root Word's 36pt indent reads as 36px
// where the page gives it 48. Measured at the reader's depth, a list wrapped
// less than Word wraps it and every page held two items too many.
let left = p.indent_left.unwrap_or(0.0).max(0.0);
if left > 0.0 {
css.push_str(&format!("--p-indent:{left:.2}pt;"));
}
// The first line's own indent, for a paragraph. NOT for a list item: there
// the hanging indent is where the bullet goes, and the text of every line,
// first included, begins at the indent proper.
let first = p.indent_first.unwrap_or(0.0).max(-left);
if first.abs() > 0.01 && p.numbering.is_none() {
css.push_str(&format!("--p-first:{first:.2}pt;"));
}
css
}
/// Whether a paragraph is a list item, and whether the list is ordered.
///
/// `bullet` is the only unordered format in OOXML; every other format is a
/// counter of some kind, so anything that is not a bullet is an ordered list.
pub fn list_kind(p: &Paragraph) -> Option {
let n = p.numbering.as_ref()?;
let ordered = !matches!(n.format.as_deref(), Some("bullet") | None);
Some(ordered)
}
/// Whether a cell is only there to be covered by a merge from the row above.
pub fn is_merged_away(cell: &Cell) -> bool {
matches!(cell.v_merge.as_deref(), Some("continue"))
}
/// A whole document: its blocks, in order.
///
/// Consecutive list paragraphs are gathered into one `
`/`` rather than
/// each becoming its own single-item list, which is what a naive block-by-block
/// walk produces and what makes such a rendering look like a stack of bullets
/// with gaps between them.
#[component]
pub fn DocxBody(
blocks: Vec,
/// Where the pages end, as `(group index, item index, the page that begins
/// there)`. The item index is [`BEFORE_GROUP`] for a break between groups
/// and the item's own index for one inside a list.
/// Worked out by [`PagedDocx`], which is the only thing that fills this in;
/// a table cell renders through here too and has no pages of its own.
#[props(default)]
marks: Vec<(usize, usize, usize, f64)>,
) -> Element {
// Group first, render second: the grouping is a property of the SEQUENCE,
// and rsx has no way to look ahead mid-iteration.
let mut groups: Vec = Vec::new();
for block in blocks {
match &block {
Block::Paragraph(p) => match list_kind(p) {
Some(ordered) => match groups.last_mut() {
Some(Group::List { ordered: o, items }) if *o == ordered => {
items.push(p.clone())
}
_ => groups.push(Group::List {
ordered,
items: vec![p.clone()],
}),
},
None => groups.push(Group::Single(block)),
},
_ => groups.push(Group::Single(block)),
}
}
// A mark can sit before a group, or BETWEEN two items of a list. Word breaks
// a page wherever the text runs out, which in these documents is usually
// in the middle of a bulleted list -- sixty of one document's sixty-eight
// paragraphs are list items -- so marking only between groups snapped every
// break back to where the list started, eight paragraphs early.
let split_at = |group: usize, item: usize| {
marks
.iter()
.find(|(g, i, _, _)| *g == group && *i == item)
.map(|(_, _, begins, spare)| (*begins, *spare))
};
rsx! {
div { class: "docx",
for (i , group) in groups.into_iter().enumerate() {
// Where a page ends, when someone has worked out where that is
// (see `PagedDocx`). Empty for a cell's contents and for a
// document nobody paginated, which is most of them.
if let Some((begins, spare)) = split_at(i, BEFORE_GROUP) {
PageMark { key: "p{i}", begins, spare }
}
match group {
Group::Single(block) => {
let rows_marked: Vec<(usize, usize, f64)> = marks
.iter()
.filter(|(g, item, _, _)| *g == i && *item != BEFORE_GROUP)
.map(|(_, item, begins, spare)| (*item, *begins, *spare))
.collect();
rsx! {
DocxBlock { key: "b{i}", block, rows_marked }
}
}
Group::List { ordered, items } => {
// A list drawn with picture bullets places itself from
// the document's own indents, so the list must not add
// its own padding on top and push it out of line with
// the paragraphs around it.
let class = match items.iter().any(|it| {
it.numbering
.as_ref()
.is_some_and(|n| n.picture().is_some())
}) {
true => "docx-list docx-list-pic",
false => "docx-list",
};
// The list, cut into runs wherever a page ends inside
// it. Each run is its own list with the mark between,
// which is what HTML has room for: a page break is not
// a list item.
// Where the run begins, if a page began with it, and the
// items in it.
#[allow(clippy::type_complexity)]
let mut runs: Vec<(Option<(usize, f64)>, Vec)> =
vec![(None, Vec::new())];
for (j, item) in items.into_iter().enumerate() {
match split_at(i, j) {
Some(mark) if j > 0 => runs.push((Some(mark), vec![item])),
_ => {
if let Some(run) = runs.last_mut() {
run.1.push(item);
}
}
}
}
// A page beginning at the list's FIRST item belongs
// before the whole list, since splitting there would
// leave an empty one. It used to be dropped instead:
// the count knew about the page and nothing on screen
// marked it, so the control offered a page it could not
// scroll to.
let ahead = split_at(i, 0);
rsx! {
if let Some((begins, spare)) = ahead {
PageMark { key: "lp{i}-first", begins, spare }
}
for (r , (mark , run)) in runs.into_iter().enumerate() {
if let Some((begins, spare)) = mark {
PageMark { key: "lp{i}-{r}", begins, spare }
}
if ordered {
ol { key: "l{i}-{r}", class,
for (j , item) in run.into_iter().enumerate() {
ListItem { key: "i{j}", item }
}
}
} else {
ul { key: "l{i}-{r}", class,
for (j , item) in run.into_iter().enumerate() {
ListItem { key: "i{j}", item }
}
}
}
}
}
}
}
}
}
}
}
/// The item index that means "before the whole group" rather than inside it.
pub const BEFORE_GROUP: usize = usize::MAX;
/// The hairline that says a page ended here, the same one the PDF reader draws.
#[component]
fn PageMark(begins: usize, #[props(default)] spare: f64) -> Element {
rsx! {
div {
class: "pdf-page-break",
role: "separator",
// The paper left under the text of the page that ends here, drawn as
// the empty space it is on the page itself.
style: "--page-spare:{spare:.0}px;",
"data-page": "{begins}",
// An ending, like the PDF's: a jump aims past it.
"data-page-ends": "true",
span { "{begins - 1}" }
}
}
}
/// A run of blocks that render together.
// One of these per block of the document, held only long enough to render it.
#[allow(clippy::large_enum_variant)]
enum Group {
Single(Block),
List {
ordered: bool,
items: Vec,
},
}
/// One cell of a table.
///
/// Its own component so the shading can be read off the cell before its content
/// moves into the body, which rsx has no room to do inline.
#[component]
fn TableCell(cell: Cell, header: bool, #[props(default)] share: Option) -> Element {
// The width Word states, for the measuring copy, and the share of the table
// it is, for anything that would rather fit the text column. Custom
// properties: the visible table sizes its columns to their contents, which
// is what keeps a table readable on a phone.
let mut shade = String::new();
if let Some(pct) = share {
shade.push_str(&format!("--c-share:{pct:.3}%;"));
}
if let Some(pt) = cell.width_pt.filter(|w| *w > 0.0) {
shade.push_str(&format!("--c-width:{pt:.2}pt;"));
}
shade.push_str(&cell_style(&cell));
let span = cell.col_span;
match header {
true => rsx! {
th { colspan: "{span}", style: "{shade}", DocxBody { blocks: cell.content } }
},
false => rsx! {
td { colspan: "{span}", style: "{shade}", DocxBody { blocks: cell.content } }
},
}
}
/// One item in a list.
///
/// Word lists can use a picture as their bullet — `numPicBullet` — and a
/// browser's own disc is not it. Where the document supplies one, the marker is
/// turned off and the picture is drawn in its place, at the size the numbering
/// definition gives it.
#[component]
fn ListItem(item: Paragraph) -> Element {
let bullet = item
.numbering
.as_ref()
.filter(|n| n.picture().is_some())
.and_then(|n| n.src.clone().map(|src| (src, n.bullet_style())));
match bullet {
Some((src, style)) => {
// Built exactly like the picture-bulleted PARAGRAPHS around it: the
// bullet inline at the head of the text, and the paragraph's own
// hanging indent placing it. That is what makes one level of
// bullets look like one level whether the document wrote them as a
// list or not — and the list's own padding is dropped, since the
// document has already said where this belongs.
let indent = paragraph_style(&item);
rsx! {
li { class: "docx-li-pic", style: "{indent}",
img { class: "docx-bullet", src: "{src}", style: "{style}", alt: "" }
{runs_of(&item)}
}
}
}
None => {
// `w:ilvl` is the outline level of the item. Every level used to
// render identically, so a document's second level read as a
// continuation of its first. Stepped and re-marked here rather than
// by nesting real lists: the parser hands these over as one flat
// run of paragraphs, and inferring a tree from the levels would
// invent structure the document did not state.
let level = item
.numbering
.as_ref()
.and_then(|n| n.level)
.unwrap_or(0)
.clamp(0, 8);
let class = match level {
0 => "docx-li",
1 => "docx-li docx-li-2",
2 => "docx-li docx-li-3",
_ => "docx-li docx-li-4",
};
// Word closes a list up: where a paragraph sets contextualSpacing,
// the space under it goes IF the next paragraph shares its style
// and sets it too. That "if" is why reading the flag off the
// document and dropping every item's spacing was wrong -- it took
// six hundred points out of a sixty-item list. The pair is what
// matters, so each item says whether it is closed up and the
// stylesheet asks about its neighbour.
let class = match item.contextual_spacing {
true => format!("{class} docx-li-tight"),
false => class.to_string(),
};
// Its own resolved typography, for the measuring copy. A list item
// used to carry none, so a page was worked out with the document's
// defaults where Word uses the item's face, size, line spacing and
// indent -- a Cambria list measured in Calibri, at a line box 8%
// too tall, half as deep as the page indents it.
let measured = measured_style(&item);
rsx! {
li { class: "{class}", style: "{measured}", {runs_of(&item)} }
}
}
}
}
/// One block: a heading, a paragraph, or a table.
#[component]
fn DocxBlock(
block: Block,
/// Page ends that fall INSIDE this block, as `(row index, the page that
/// begins there)`. Only a table has anywhere inside it to put one: Word
/// carries a table's rows onto the next page, and a document that is mostly
/// table has almost nowhere else a page can end.
#[props(default)]
rows_marked: Vec<(usize, usize, f64)>,
) -> Element {
match block {
Block::Paragraph(p) => {
let style = paragraph_style(&p);
let inner = runs_of(&p);
// A dynamic tag name is not expressible in rsx, so the six headings
// are written out. HTML has six; deeper levels were clamped when the
// level was worked out.
match heading_level(p.style_id.as_deref(), p.outline_level) {
Some(1) => {
rsx! { h1 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
Some(2) => {
rsx! { h2 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
Some(3) => {
rsx! { h3 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
Some(4) => {
rsx! { h4 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
Some(5) => {
rsx! { h5 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
Some(_) => {
rsx! { h6 { class: "docx-h", style: "{style}", "data-keep-next": "{p.keep_next}", {inner} } }
}
// An empty paragraph is a deliberate blank line in Word, so it
// keeps its element rather than being dropped.
None if p.keep_next => rsx! {
p { class: "docx-p", style: "{style}", "data-keep-next": "true", {inner} }
},
None => rsx! { p { class: "docx-p", style: "{style}", {inner} } },
}
}
Block::Table(t) => {
// How wide the mark's own row has to be to span the table.
let across = t.rows.iter().map(|r| r.cells.len()).max().unwrap_or(1);
// Word's column widths, and the SUM of them, for the off-screen copy
// to wrap the text where Word wraps it.
//
// Left to itself the browser fits columns to their content, which
// is right on screen and nothing like Word: it gave this table
// 159/81/402 where Word gives 290/75/305.
//
// The sum is the part that took two goes to get right. This
// document's three columns come to 670px against a 642px text
// column, because **Word lets a table run into the margin**; a table
// told only what its columns are is still capped at the box around
// it, and the browser scales them down to fit — 290/75/305 became
// 275/76/290, five per cent narrower, and five per cent narrower
// wraps a seven-line cell into eight. That is one line per row, and
// over one document's tables it came to the better part of a page,
// which moved a break a row late. So the table is told its own width
// as well, and allowed to overflow exactly as it does in Word.
//
// Shares alongside, for the reading copy, which would rather fit
// the text column than reproduce the page.
let widths: Vec = t
.rows
.iter()
.max_by_key(|r| r.cells.len())
.map(|r| r.cells.iter().map(|c| c.width_pt.unwrap_or(0.0)).collect())
.unwrap_or_default();
let stated: f64 = widths.iter().sum();
let shares: Vec