Browse Source

Revamp Docs DOCX fidelity and browser PDF export

Overhauls Office DOCX import/export to preserve rich layout and formatting, including style inheritance, numbering, tabs, line spacing, borders, table geometry, header/footer HTML, footnotes, page fields, image orientation/cropping, and page setup distances. Adds new parsing/writer helpers and broad test coverage for rich round-trips and fidelity edge cases.

On the web side, introduces a shared PDF core/drawing stack and moves Docs PDF generation to browser rendering with worker-based assembly, plus a new Docs layout engine for pagination, split handling, tab/list metrics, footnotes, page sheets, and header/footer/page-number rendering aligned with the live editor.
Toby Chui 2 weeks ago
parent
commit
a7aabeccdf

+ 19 - 1
src/mod/office/docx.go

@@ -21,10 +21,24 @@ import (
 type Document struct {
 type Document struct {
 	HTML        string    `json:"html"`
 	HTML        string    `json:"html"`
 	Page        *PageConf `json:"page,omitempty"`
 	Page        *PageConf `json:"page,omitempty"`
-	Header      string    `json:"header,omitempty"`
+	Header      string    `json:"header,omitempty"` // plain text (older documents)
 	Footer      string    `json:"footer,omitempty"`
 	Footer      string    `json:"footer,omitempty"`
+	HeaderHTML  string    `json:"headerHtml,omitempty"` // rich header; wins over Header
+	FooterHTML  string    `json:"footerHtml,omitempty"`
 	PageNumbers bool      `json:"pageNumbers,omitempty"`
 	PageNumbers bool      `json:"pageNumbers,omitempty"`
 	HFMode      string    `json:"hfMode,omitempty"` // header/footer repetition
 	HFMode      string    `json:"hfMode,omitempty"` // header/footer repetition
+	// Footnotes are referenced from the text by <sup class="doc-fnref"
+	// data-fn="ID">; numbered in reference order
+	Footnotes []Footnote `json:"footnotes,omitempty"`
+	// LineSpacing is the multiple of single spacing a paragraph without
+	// data-ls uses (0 = the editor's default)
+	LineSpacing float64 `json:"lineSpacing,omitempty"`
+}
+
+// Footnote is one footnote's content (block HTML)
+type Footnote struct {
+	ID   string `json:"id"`
+	HTML string `json:"html"`
 }
 }
 
 
 // Header/footer repetition modes (body.hfMode); the empty string means
 // Header/footer repetition modes (body.hfMode); the empty string means
@@ -61,6 +75,10 @@ type PageConf struct {
 	Margins     *MarginsMM `json:"margins,omitempty"`
 	Margins     *MarginsMM `json:"margins,omitempty"`
 	Columns     int        `json:"columns,omitempty"` // text columns (0/1 = single)
 	Columns     int        `json:"columns,omitempty"` // text columns (0/1 = single)
 	ColGap      float64    `json:"colGap,omitempty"`  // gap between columns, mm
 	ColGap      float64    `json:"colGap,omitempty"`  // gap between columns, mm
+	// distance of the header from the top edge / the footer from the
+	// bottom edge, mm (nil = the editor's default)
+	HeaderDist *float64 `json:"headerDist,omitempty"`
+	FooterDist *float64 `json:"footerDist,omitempty"`
 }
 }
 
 
 // MarginsMM holds page margins in millimetres
 // MarginsMM holds page margins in millimetres

+ 40 - 23
src/mod/office/docx_fidelity_test.go

@@ -7,6 +7,7 @@ import (
 	"image"
 	"image"
 	"image/color"
 	"image/color"
 	"image/png"
 	"image/png"
+	"math"
 	"strings"
 	"strings"
 	"testing"
 	"testing"
 )
 )
@@ -61,15 +62,28 @@ func TestDocxImageNoSizeUsesNatural(t *testing.T) {
 
 
 func TestDocxImageCappedToTextWidth(t *testing.T) {
 func TestDocxImageCappedToTextWidth(t *testing.T) {
 	src := makePngDataURL(t, 100, 50)
 	src := makePngDataURL(t, 100, 50)
-	doc := &Document{HTML: `<p><img src="` + src + `" width="1240"></p>`}
-	data, err := BuildDocx(doc)
-	if err != nil {
-		t.Fatalf("BuildDocx: %v", err)
-	}
-	body := string(zipPart(t, data, "word/document.xml"))
-	want := fmt.Sprintf(`<wp:extent cx="%d" cy="%d"/>`, pxToEmu(620), pxToEmu(310))
-	if !strings.Contains(body, want) {
-		t.Errorf("expected capped %s, got: %s", want, snippetAround(body, "wp:extent"))
+	textW := textWidthPt(nil) // A4, 25.4mm margins
+	emu := func(pt float64) int64 { return int64(math.Round(pt * 12700)) }
+	tests := []struct {
+		name, img string
+		cx, cy    int64
+	}{
+		// no stated height: the editor scales it with the width
+		{"automatic height", `<img src="` + src + `" width="1240">`, emu(textW), emu(textW / 2)},
+		// a stated height stays as the editor shows it (max-width only)
+		{"stated height", `<img src="` + src + `" style="width:600pt;height:200pt;">`, emu(textW), emu(200)},
+		{"narrow picture", `<img src="` + src + `" style="width:300pt;height:150pt;">`, emu(300), emu(150)},
+	}
+	for _, tc := range tests {
+		data, err := BuildDocx(&Document{HTML: `<p>` + tc.img + `</p>`})
+		if err != nil {
+			t.Fatalf("%s: BuildDocx: %v", tc.name, err)
+		}
+		body := string(zipPart(t, data, "word/document.xml"))
+		want := fmt.Sprintf(`<wp:extent cx="%d" cy="%d"/>`, tc.cx, tc.cy)
+		if !strings.Contains(body, want) {
+			t.Errorf("%s: expected %s, got: %s", tc.name, want, snippetAround(body, "wp:extent"))
+		}
 	}
 	}
 }
 }
 
 
@@ -84,27 +98,30 @@ func TestDocxTableWidthAndShading(t *testing.T) {
 		t.Fatalf("BuildDocx: %v", err)
 		t.Fatalf("BuildDocx: %v", err)
 	}
 	}
 	body := string(zipPart(t, data, "word/document.xml"))
 	body := string(zipPart(t, data, "word/document.xml"))
-	// full-width fixed layout
-	if !strings.Contains(body, `<w:tblW w:w="5000" w:type="pct"/>`) {
-		t.Error("table is not full width (pct)")
+	// full text width (A4, 25.4mm margins: 9026 twips), fixed layout
+	if !strings.Contains(body, `<w:tblW w:w="9026" w:type="dxa"/>`) {
+		t.Errorf("table is not full width: %s", snippetAround(body, "tblW"))
 	}
 	}
 	if !strings.Contains(body, `<w:tblLayout w:type="fixed"/>`) {
 	if !strings.Contains(body, `<w:tblLayout w:type="fixed"/>`) {
 		t.Error("table layout is not fixed")
 		t.Error("table layout is not fixed")
 	}
 	}
 	// column proportions from the colgroup: 60% and 40% of 9026 twips
 	// column proportions from the colgroup: 60% and 40% of 9026 twips
-	if !strings.Contains(body, `<w:gridCol w:w="5415"/>`) ||
+	if !strings.Contains(body, `<w:gridCol w:w="5416"/>`) ||
 		!strings.Contains(body, `<w:gridCol w:w="3610"/>`) {
 		!strings.Contains(body, `<w:gridCol w:w="3610"/>`) {
 		t.Errorf("grid columns do not follow the colgroup: %s", snippetAround(body, "tblGrid"))
 		t.Errorf("grid columns do not follow the colgroup: %s", snippetAround(body, "tblGrid"))
 	}
 	}
-	// per-cell pct widths (fiftieths of a percent)
-	if !strings.Contains(body, `<w:tcW w:w="3000" w:type="pct"/>`) ||
-		!strings.Contains(body, `<w:tcW w:w="2000" w:type="pct"/>`) {
+	if !strings.Contains(body, `<w:tcW w:w="5416" w:type="dxa"/>`) ||
+		!strings.Contains(body, `<w:tcW w:w="3610" w:type="dxa"/>`) {
 		t.Errorf("cell widths not proportional: %s", snippetAround(body, "tcW"))
 		t.Errorf("cell widths not proportional: %s", snippetAround(body, "tcW"))
 	}
 	}
 	// theme shading + bold survive
 	// theme shading + bold survive
 	if !strings.Contains(body, `<w:shd w:val="clear" w:color="auto" w:fill="3C4043"/>`) {
 	if !strings.Contains(body, `<w:shd w:val="clear" w:color="auto" w:fill="3C4043"/>`) {
 		t.Errorf("cell shading lost: %s", snippetAround(body, "shd"))
 		t.Errorf("cell shading lost: %s", snippetAround(body, "shd"))
 	}
 	}
+	// ... on the cell alone: the runs inside do not shade themselves again
+	if n := strings.Count(body, `w:fill="3C4043"`); n != 1 {
+		t.Errorf("cell shading written %d times, want once (tcPr): %s", n, body)
+	}
 }
 }
 
 
 func TestDocxTableWidthRoundTrip(t *testing.T) {
 func TestDocxTableWidthRoundTrip(t *testing.T) {
@@ -117,12 +134,13 @@ func TestDocxTableWidthRoundTrip(t *testing.T) {
 		t.Fatalf("BuildDocx: %v", err)
 		t.Fatalf("BuildDocx: %v", err)
 	}
 	}
 	body := string(zipPart(t, data, "word/document.xml"))
 	body := string(zipPart(t, data, "word/document.xml"))
-	// 60% of the text column -> tblW 3000 pct
-	if !strings.Contains(body, `<w:tblW w:w="3000" w:type="pct"/>`) {
-		t.Errorf("table width not 60 pct: %s", snippetAround(body, "tblW"))
+	// 372px = 279pt = 5580 twips
+	if !strings.Contains(body, `<w:tblW w:w="5580" w:type="dxa"/>`) {
+		t.Errorf("table width not 279pt: %s", snippetAround(body, "tblW"))
 	}
 	}
 	// px colgroup ratios (50/25/25) scaled into the grid
 	// px colgroup ratios (50/25/25) scaled into the grid
-	if !strings.Contains(body, `<w:tcW w:w="2500" w:type="pct"/>`) {
+	if !strings.Contains(body, `<w:tcW w:w="2790" w:type="dxa"/>`) ||
+		!strings.Contains(body, `<w:tcW w:w="1395" w:type="dxa"/>`) {
 		t.Errorf("cell widths not 50/25/25: %s", snippetAround(body, "tcW"))
 		t.Errorf("cell widths not 50/25/25: %s", snippetAround(body, "tcW"))
 	}
 	}
 	back, err := ParseDocx(data)
 	back, err := ParseDocx(data)
@@ -132,11 +150,10 @@ func TestDocxTableWidthRoundTrip(t *testing.T) {
 	if !strings.Contains(back.HTML, `class="of-table"`) {
 	if !strings.Contains(back.HTML, `class="of-table"`) {
 		t.Errorf("imported table lost the of-table class: %s", back.HTML)
 		t.Errorf("imported table lost the of-table class: %s", back.HTML)
 	}
 	}
-	if !strings.Contains(back.HTML, "width:60%") {
+	if !strings.Contains(back.HTML, "width:279pt") {
 		t.Errorf("imported table lost its width: %s", back.HTML)
 		t.Errorf("imported table lost its width: %s", back.HTML)
 	}
 	}
-	// twip rounding may give 50.01% - the proportion is what matters
-	if !strings.Contains(back.HTML, "<colgroup>") || !strings.Contains(back.HTML, "width:50") {
+	if !strings.Contains(back.HTML, `<colgroup><col style="width:139.5pt"><col style="width:69.75pt"><col style="width:69.75pt"></colgroup>`) {
 		t.Errorf("imported table lost column proportions: %s", back.HTML)
 		t.Errorf("imported table lost column proportions: %s", back.HTML)
 	}
 	}
 }
 }

+ 245 - 0
src/mod/office/docx_numbering.go

@@ -0,0 +1,245 @@
+package office
+
+/*
+	docx_numbering.go - list numbering definitions (numbering.xml) and the
+	counters that turn them into marker text.
+
+	A Word list paragraph only says "numId 6, level 1". What that looks like
+	- "b.", "ii)", "1.2.", a bullet glyph - and where the marker and the text
+	sit lives in the abstract numbering definition the numId points at,
+	possibly overridden per numId. The counters belong to the numId and run
+	through the whole document, so a list interrupted by a paragraph carries
+	on at the next number.
+*/
+
+import (
+	"strconv"
+	"strings"
+)
+
+type docxNumLevel struct {
+	fmt      string // decimal, lowerLetter, upperLetter, lowerRoman, upperRoman, bullet, none ...
+	text     string // lvlText, e.g. "%1." or a bullet glyph
+	start    int
+	indL     optNum // twips
+	indHang  optNum
+	indFirst optNum
+	rPr      docxRPr
+	restart  optNum // lvlRestart
+}
+
+type docxNumDefs struct {
+	abstract map[string]map[int]*docxNumLevel // abstractNumId -> levels
+	nums     map[string]string                // numId -> abstractNumId
+	override map[string]map[int]*docxNumLevel // numId -> overridden levels
+	startOvr map[string]map[int]int           // numId -> startOverride
+	counters map[string][]int                 // numId -> current value per level
+	started  map[string][]bool
+}
+
+func parseNumbering(raw []byte) *docxNumDefs {
+	nb := &docxNumDefs{
+		abstract: map[string]map[int]*docxNumLevel{},
+		nums:     map[string]string{},
+		override: map[string]map[int]*docxNumLevel{},
+		startOvr: map[string]map[int]int{},
+		counters: map[string][]int{},
+		started:  map[string][]bool{},
+	}
+	if raw == nil {
+		return nb
+	}
+	tree, err := parseXMLTree(raw)
+	if err != nil {
+		return nb
+	}
+	for _, an := range tree.all("abstractNum") {
+		levels := map[int]*docxNumLevel{}
+		for _, lvl := range an.all("lvl") {
+			idx, err := strconv.Atoi(lvl.attr("ilvl"))
+			if err != nil {
+				continue
+			}
+			levels[idx] = parseNumLevel(lvl)
+		}
+		nb.abstract[an.attr("abstractNumId")] = levels
+	}
+	for _, num := range tree.all("num") {
+		id := num.attr("numId")
+		if ref := num.first("abstractNumId"); ref != nil {
+			nb.nums[id] = ref.attr("val")
+		}
+		for _, ov := range num.all("lvlOverride") {
+			idx, err := strconv.Atoi(ov.attr("ilvl"))
+			if err != nil {
+				continue
+			}
+			if so := ov.first("startOverride"); so != nil {
+				if v, err := strconv.Atoi(so.attr("val")); err == nil {
+					if nb.startOvr[id] == nil {
+						nb.startOvr[id] = map[int]int{}
+					}
+					nb.startOvr[id][idx] = v
+				}
+			}
+			if lvl := ov.first("lvl"); lvl != nil {
+				if nb.override[id] == nil {
+					nb.override[id] = map[int]*docxNumLevel{}
+				}
+				nb.override[id][idx] = parseNumLevel(lvl)
+			}
+		}
+	}
+	return nb
+}
+
+func parseNumLevel(lvl *xnode) *docxNumLevel {
+	l := &docxNumLevel{fmt: "decimal", start: 1}
+	if f := lvl.first("numFmt"); f != nil {
+		l.fmt = f.attr("val")
+	}
+	if t := lvl.first("lvlText"); t != nil {
+		l.text = t.attr("val")
+	}
+	if s := lvl.first("start"); s != nil {
+		if v, err := strconv.Atoi(s.attr("val")); err == nil {
+			l.start = v
+		}
+	}
+	l.restart = numAttr(lvl.first("lvlRestart"), "val")
+	if ppr := lvl.first("pPr"); ppr != nil {
+		p := parsePPr(ppr)
+		l.indL, l.indHang, l.indFirst = p.indL, p.indHang, p.indFirst
+	}
+	l.rPr = parseRPr(lvl.first("rPr"))
+	return l
+}
+
+// level returns the definition of one level of a list instance
+func (nb *docxNumDefs) level(numID string, ilvl int) *docxNumLevel {
+	if ov, ok := nb.override[numID]; ok {
+		if l, ok := ov[ilvl]; ok {
+			return l
+		}
+	}
+	if levels, ok := nb.abstract[nb.nums[numID]]; ok {
+		if l, ok := levels[ilvl]; ok {
+			return l
+		}
+	}
+	return nil
+}
+
+// exists reports whether numID names a real list (numId 0 switches
+// numbering off for a paragraph whose style would otherwise number it)
+func (nb *docxNumDefs) exists(numID string) bool {
+	if numID == "" || numID == "0" {
+		return false
+	}
+	_, ok := nb.nums[numID]
+	return ok
+}
+
+// next advances the counter for an item at ilvl and returns its value
+func (nb *docxNumDefs) next(numID string, ilvl int) int {
+	c := nb.counters[numID]
+	st := nb.started[numID]
+	if c == nil {
+		c = make([]int, 9)
+		st = make([]bool, 9)
+	}
+	if ilvl < 0 {
+		ilvl = 0
+	}
+	if ilvl > 8 {
+		ilvl = 8
+	}
+	startOf := func(l int) int {
+		if so, ok := nb.startOvr[numID][l]; ok {
+			return so
+		}
+		if def := nb.level(numID, l); def != nil {
+			return def.start
+		}
+		return 1
+	}
+	if !st[ilvl] {
+		c[ilvl] = startOf(ilvl)
+		st[ilvl] = true
+	} else {
+		c[ilvl]++
+	}
+	// an item restarts every deeper level
+	for l := ilvl + 1; l < 9; l++ {
+		st[l] = false
+	}
+	// shallower levels that never had an item count as their start value
+	for l := 0; l < ilvl; l++ {
+		if !st[l] {
+			c[l] = startOf(l)
+		}
+	}
+	nb.counters[numID] = c
+	nb.started[numID] = st
+	return c[ilvl]
+}
+
+// htmlListFormat maps a Word number format onto the editor's list format
+// names (the same vocabulary docs_layout.js understands)
+func htmlListFormat(f string) string {
+	switch f {
+	case "bullet", "decimal", "lowerLetter", "upperLetter", "lowerRoman", "upperRoman", "none":
+		return f
+	case "decimalZero":
+		return "decimalZero"
+	}
+	return "decimal"
+}
+
+// formatListNumber renders one counter value in a Word number format
+func formatListNumber(v int, f string) string {
+	switch f {
+	case "lowerLetter", "upperLetter":
+		s := alphaNumber(v)
+		if f == "upperLetter" {
+			return strings.ToUpper(s)
+		}
+		return s
+	case "lowerRoman":
+		return strings.ToLower(romanNumber(v))
+	case "upperRoman":
+		return romanNumber(v)
+	case "decimalZero":
+		if v < 10 {
+			return "0" + strconv.Itoa(v)
+		}
+	case "none", "bullet":
+		return ""
+	}
+	return strconv.Itoa(v)
+}
+
+// alphaNumber: 1 -> a, 26 -> z, 27 -> aa (Word repeats the letter)
+func alphaNumber(v int) string {
+	if v < 1 {
+		return ""
+	}
+	letter := string(rune('a' + (v-1)%26))
+	return strings.Repeat(letter, (v-1)/26+1)
+}
+
+func romanNumber(v int) string {
+	if v < 1 || v > 3999 {
+		return strconv.Itoa(v)
+	}
+	vals := []int{1000, 900, 500, 400, 100, 90, 50, 40, 10, 9, 5, 4, 1}
+	syms := []string{"M", "CM", "D", "CD", "C", "XC", "L", "XL", "X", "IX", "V", "IV", "I"}
+	var sb strings.Builder
+	for i, n := range vals {
+		for v >= n {
+			sb.WriteString(syms[i])
+			v -= n
+		}
+	}
+	return sb.String()
+}

+ 9 - 6
src/mod/office/docx_pagebreak_test.go

@@ -79,13 +79,16 @@ func TestDocxHeaderFooterNotCentered(t *testing.T) {
 		if !strings.Contains(raw, tc.want) {
 		if !strings.Contains(raw, tc.want) {
 			t.Errorf("%s missing its text: %s", tc.part, raw)
 			t.Errorf("%s missing its text: %s", tc.part, raw)
 		}
 		}
-		// the editor renders header/footer left aligned - the export must match
-		if strings.Contains(raw, `<w:jc w:val="center"/>`) {
-			t.Errorf("%s is centred but the editor left aligns it: %s", tc.part, raw)
+		// the editor renders header/footer text left aligned - the export
+		// must match (only the page number line below is centred)
+		text := raw[:strings.Index(raw, tc.want)]
+		if strings.Contains(text[strings.LastIndex(text, "<w:p>"):], `<w:jc w:val="center"/>`) {
+			t.Errorf("%s text is centred but the editor left aligns it: %s", tc.part, raw)
 		}
 		}
 	}
 	}
-	// the PAGE field must survive the alignment fix
-	if !strings.Contains(string(zipPart(t, data, "word/footer1.xml")), "PAGE") {
-		t.Error("footer lost its page number field")
+	// the page number the editor centres under the text survives
+	footer := string(zipPart(t, data, "word/footer1.xml"))
+	if !strings.Contains(footer, `<w:pStyle w:val="ArozPageNumber"/><w:jc w:val="center"/></w:pPr><w:fldSimple w:instr=" PAGE ">`) {
+		t.Errorf("footer lost its page number field: %s", footer)
 	}
 	}
 }
 }

+ 108 - 0
src/mod/office/docx_picture.go

@@ -0,0 +1,108 @@
+package office
+
+/*
+	docx_picture.go - turning and mirroring a picture's bitmap
+
+	A DrawingML picture may be rotated by quarter turns or flipped. The
+	editor has no notion of a turned picture (a CSS transform would not move
+	the text around it), so the import bakes the turn into the bitmap and
+	swaps the frame: the picture then lays out, prints and saves as it looks.
+*/
+
+import (
+	"bytes"
+	"image"
+	"image/draw"
+	_ "image/gif" // decoding of GIF pictures
+	"image/jpeg"
+	"image/png"
+	"math"
+)
+
+// pictureTurns reads a DrawingML rotation (60000ths of a degree, clockwise)
+// as a count of quarter turns; ok is false for any other angle
+func pictureTurns(rot float64) (int, bool) {
+	deg := rot / 60000
+	q := math.Round(deg / 90)
+	if math.Abs(deg-q*90) > 1 {
+		return 0, false
+	}
+	return ((int(q) % 4) + 4) % 4, true
+}
+
+// orientPixels mirrors (first) and then turns an image clockwise
+func orientPixels(src image.Image, turns int, flipH, flipV bool) *image.NRGBA {
+	b := src.Bounds()
+	w, h := b.Dx(), b.Dy()
+	in := image.NewNRGBA(image.Rect(0, 0, w, h))
+	draw.Draw(in, in.Bounds(), src, b.Min, draw.Src)
+	ow, oh := w, h
+	if turns%2 == 1 {
+		ow, oh = h, w
+	}
+	out := image.NewNRGBA(image.Rect(0, 0, ow, oh))
+	for y := 0; y < h; y++ {
+		for x := 0; x < w; x++ {
+			sx, sy := x, y
+			if flipH {
+				sx = w - 1 - x
+			}
+			if flipV {
+				sy = h - 1 - y
+			}
+			dx, dy := x, y
+			switch turns {
+			case 1:
+				dx, dy = h-1-y, x
+			case 2:
+				dx, dy = w-1-x, h-1-y
+			case 3:
+				dx, dy = y, w-1-x
+			}
+			si := in.PixOffset(sx, sy)
+			di := out.PixOffset(dx, dy)
+			copy(out.Pix[di:di+4], in.Pix[si:si+4])
+		}
+	}
+	return out
+}
+
+// orientPicture re-encodes a picture turned and mirrored as DrawingML asks;
+// ok is false when the format cannot be decoded (the picture stays as is)
+func orientPicture(data []byte, ext string, turns int, flipH, flipV bool) ([]byte, string, bool) {
+	if turns == 0 && !flipH && !flipV {
+		return data, ext, true
+	}
+	img, format, err := image.Decode(bytes.NewReader(data))
+	if err != nil {
+		return data, ext, false
+	}
+	out := orientPixels(img, turns, flipH, flipV)
+	var buf bytes.Buffer
+	if format == "jpeg" {
+		if err := jpeg.Encode(&buf, out, &jpeg.Options{Quality: 92}); err != nil {
+			return data, ext, false
+		}
+		return buf.Bytes(), "jpeg", true
+	}
+	if err := png.Encode(&buf, out); err != nil {
+		return data, ext, false
+	}
+	return buf.Bytes(), "png", true
+}
+
+// orientInsets moves a crop (top, right, bottom, left) with the bitmap
+func orientInsets(in [4]float64, turns int, flipH, flipV bool) [4]float64 {
+	t, r, b, l := in[0], in[1], in[2], in[3]
+	if flipH {
+		l, r = r, l
+	}
+	if flipV {
+		t, b = b, t
+	}
+	for i := 0; i < turns; i++ {
+		// a clockwise turn: the left edge becomes the top
+		t, r, b, l = l, t, r, b
+	}
+	return [4]float64{t, r, b, l}
+}

+ 662 - 0
src/mod/office/docx_props.go

@@ -0,0 +1,662 @@
+package office
+
+/*
+	docx_props.go - WordprocessingML formatting properties and the style
+	inheritance that resolves them.
+
+	A paragraph's appearance in Word is almost never stated on the paragraph
+	itself. It is the sum, lowest priority first, of
+
+	    docDefaults (rPrDefault / pPrDefault)
+	    the paragraph style, walked up its basedOn chain
+	    the paragraph's own pPr
+	    for a run: the character style chain, then the run's own rPr
+
+	and a table adds its table style (and TableNormal behind that) for
+	borders and cell margins. Every property below is therefore optional -
+	"not stated" and "stated as off" are different things, and the second
+	one has to win over an inherited "on". Google Docs writes explicit
+	w:val="0" for nearly every toggle, which is exactly the case an
+	"element present = on" reader gets wrong (italic headings, bold body).
+*/
+
+import (
+	"strconv"
+	"strings"
+)
+
+// optBool is a toggle that may or may not be stated
+type optBool struct {
+	set bool
+	v   bool
+}
+
+// optNum is a number that may or may not be stated
+type optNum struct {
+	set bool
+	v   float64
+}
+
+func (o *optBool) merge(src optBool) {
+	if src.set {
+		*o = src
+	}
+}
+
+func (o *optNum) merge(src optNum) {
+	if src.set {
+		*o = src
+	}
+}
+
+// onOff reads a WordprocessingML ST_OnOff element: present without w:val
+// means on
+func onOff(n *xnode) optBool {
+	if n == nil {
+		return optBool{}
+	}
+	switch strings.ToLower(n.attr("val")) {
+	case "0", "false", "off", "none":
+		return optBool{set: true, v: false}
+	}
+	return optBool{set: true, v: true}
+}
+
+// numAttr reads a numeric attribute (Google Docs writes "100.0")
+func numAttr(n *xnode, name string) optNum {
+	if n == nil {
+		return optNum{}
+	}
+	s := strings.TrimSpace(n.attr(name))
+	if s == "" {
+		return optNum{}
+	}
+	v, err := strconv.ParseFloat(s, 64)
+	if err != nil {
+		return optNum{}
+	}
+	return optNum{set: true, v: v}
+}
+
+/* ---------------- run properties ---------------- */
+
+type docxRPr struct {
+	b, i, strike, dstrike, caps, smallCaps, vanish optBool
+	u                                              string // "" = not stated, "none" = off
+	color                                          string // RRGGBB, "auto", or ""
+	highlight                                      string // named highlight colour
+	shd                                            string // RRGGBB fill
+	sz                                             optNum // half-points
+	fontASCII, fontEA                              string
+	vert                                           string // superscript | subscript | baseline
+	rStyle                                         string
+}
+
+func parseRPr(n *xnode) docxRPr {
+	r := docxRPr{}
+	if n == nil {
+		return r
+	}
+	r.b = onOff(n.first("b"))
+	r.i = onOff(n.first("i"))
+	r.strike = onOff(n.first("strike"))
+	r.dstrike = onOff(n.first("dstrike"))
+	r.caps = onOff(n.first("caps"))
+	r.smallCaps = onOff(n.first("smallCaps"))
+	r.vanish = onOff(n.first("vanish"))
+	if u := n.first("u"); u != nil {
+		r.u = u.attr("val")
+		if r.u == "" {
+			r.u = "single"
+		}
+	}
+	if c := n.first("color"); c != nil {
+		r.color = strings.ToUpper(c.attr("val"))
+	}
+	if h := n.first("highlight"); h != nil {
+		r.highlight = h.attr("val")
+	}
+	if s := n.first("shd"); s != nil {
+		if f := strings.ToUpper(s.attr("fill")); len(f) == 6 {
+			r.shd = f
+		} else if f == "AUTO" {
+			r.shd = "auto"
+		}
+	}
+	r.sz = numAttr(n.first("sz"), "val")
+	if f := n.first("rFonts"); f != nil {
+		r.fontASCII = f.attr("ascii")
+		if r.fontASCII == "" {
+			r.fontASCII = f.attr("hAnsi")
+		}
+		r.fontEA = f.attr("eastAsia")
+	}
+	if va := n.first("vertAlign"); va != nil {
+		r.vert = va.attr("val")
+	}
+	if rs := n.first("rStyle"); rs != nil {
+		r.rStyle = rs.attr("val")
+	}
+	return r
+}
+
+func (r *docxRPr) merge(src docxRPr) {
+	r.b.merge(src.b)
+	r.i.merge(src.i)
+	r.strike.merge(src.strike)
+	r.dstrike.merge(src.dstrike)
+	r.caps.merge(src.caps)
+	r.smallCaps.merge(src.smallCaps)
+	r.vanish.merge(src.vanish)
+	if src.u != "" {
+		r.u = src.u
+	}
+	if src.color != "" {
+		r.color = src.color
+	}
+	if src.highlight != "" {
+		r.highlight = src.highlight
+	}
+	if src.shd != "" {
+		r.shd = src.shd
+	}
+	r.sz.merge(src.sz)
+	if src.fontASCII != "" {
+		r.fontASCII = src.fontASCII
+	}
+	if src.fontEA != "" {
+		r.fontEA = src.fontEA
+	}
+	if src.vert != "" {
+		r.vert = src.vert
+	}
+}
+
+/* ---------------- paragraph properties ---------------- */
+
+type docxBorder struct {
+	val   string
+	sz    float64 // eighths of a point
+	color string
+	space float64 // points
+}
+
+// visible reports whether the border draws anything
+func (b docxBorder) visible() bool {
+	switch b.val {
+	case "", "nil", "none":
+		return false
+	}
+	return true
+}
+
+func parseBorder(n *xnode) (docxBorder, bool) {
+	if n == nil {
+		return docxBorder{}, false
+	}
+	b := docxBorder{val: n.attr("val"), color: strings.ToUpper(n.attr("color"))}
+	if v := numAttr(n, "sz"); v.set {
+		b.sz = v.v
+	}
+	if v := numAttr(n, "space"); v.set {
+		b.space = v.v
+	}
+	return b, true
+}
+
+type docxTab struct {
+	align  string  // left | right | center | decimal | clear
+	pos    float64 // twips from the text column's left edge
+	leader string  // none | dot | hyphen | underscore ...
+}
+
+type docxPPr struct {
+	style                                string
+	before, after, line                  optNum
+	lineRule                             string
+	indL, indR, indFirst, indHang        optNum
+	jc                                   string
+	keepNext, keepLines, pageBreakBefore optBool
+	widow, contextual                    optBool
+	shd                                  string
+	borders                              map[string]docxBorder
+	numID                                string
+	numSet                               bool
+	ilvl                                 optNum
+	tabs                                 []docxTab
+	sect                                 *xnode
+	mark                                 docxRPr // paragraph mark run properties
+	outline                              optNum
+}
+
+func parsePPr(n *xnode) docxPPr {
+	p := docxPPr{}
+	if n == nil {
+		return p
+	}
+	if ps := n.first("pStyle"); ps != nil {
+		p.style = ps.attr("val")
+	}
+	if sp := n.first("spacing"); sp != nil {
+		p.before = numAttr(sp, "before")
+		p.after = numAttr(sp, "after")
+		p.line = numAttr(sp, "line")
+		p.lineRule = sp.attr("lineRule")
+		// "auto" spacing before/after is Word's HTML-ish 14pt; close enough
+		if onOff2(sp.attr("beforeAutospacing")) {
+			p.before = optNum{set: true, v: 280}
+		}
+		if onOff2(sp.attr("afterAutospacing")) {
+			p.after = optNum{set: true, v: 280}
+		}
+	}
+	if ind := n.first("ind"); ind != nil {
+		p.indL = numAttr(ind, "left")
+		if !p.indL.set {
+			p.indL = numAttr(ind, "start")
+		}
+		p.indR = numAttr(ind, "right")
+		if !p.indR.set {
+			p.indR = numAttr(ind, "end")
+		}
+		p.indFirst = numAttr(ind, "firstLine")
+		p.indHang = numAttr(ind, "hanging")
+		// a stated firstLine clears an inherited hanging indent and vice versa
+		if p.indFirst.set && !p.indHang.set {
+			p.indHang = optNum{set: true, v: 0}
+		}
+		if p.indHang.set && !p.indFirst.set {
+			p.indFirst = optNum{set: true, v: 0}
+		}
+	}
+	if jc := n.first("jc"); jc != nil {
+		p.jc = jc.attr("val")
+	}
+	p.keepNext = onOff(n.first("keepNext"))
+	p.keepLines = onOff(n.first("keepLines"))
+	p.pageBreakBefore = onOff(n.first("pageBreakBefore"))
+	p.widow = onOff(n.first("widowControl"))
+	p.contextual = onOff(n.first("contextualSpacing"))
+	if s := n.first("shd"); s != nil {
+		if f := strings.ToUpper(s.attr("fill")); len(f) == 6 {
+			p.shd = f
+		} else if f == "AUTO" {
+			p.shd = "auto"
+		}
+	}
+	if bd := n.first("pBdr"); bd != nil {
+		p.borders = map[string]docxBorder{}
+		for _, side := range []string{"top", "left", "bottom", "right"} {
+			if b, ok := parseBorder(bd.first(side)); ok {
+				p.borders[side] = b
+			}
+		}
+	}
+	if np := n.first("numPr"); np != nil {
+		if id := np.first("numId"); id != nil {
+			p.numID = id.attr("val")
+			p.numSet = true
+		}
+		p.ilvl = numAttr(np.first("ilvl"), "val")
+	}
+	if tabs := n.first("tabs"); tabs != nil {
+		for _, t := range tabs.all("tab") {
+			pos := numAttr(t, "pos")
+			if !pos.set {
+				continue
+			}
+			p.tabs = append(p.tabs, docxTab{align: t.attr("val"), pos: pos.v, leader: t.attr("leader")})
+		}
+	}
+	p.sect = n.first("sectPr")
+	p.mark = parseRPr(n.first("rPr"))
+	p.outline = numAttr(n.first("outlineLvl"), "val")
+	return p
+}
+
+func onOff2(s string) bool {
+	switch strings.ToLower(s) {
+	case "1", "true", "on":
+		return true
+	}
+	return false
+}
+
+func (p *docxPPr) merge(src docxPPr) {
+	if src.style != "" {
+		p.style = src.style
+	}
+	p.before.merge(src.before)
+	p.after.merge(src.after)
+	if src.line.set {
+		p.line = src.line
+		p.lineRule = src.lineRule
+	}
+	p.indL.merge(src.indL)
+	p.indR.merge(src.indR)
+	p.indFirst.merge(src.indFirst)
+	p.indHang.merge(src.indHang)
+	if src.jc != "" {
+		p.jc = src.jc
+	}
+	p.keepNext.merge(src.keepNext)
+	p.keepLines.merge(src.keepLines)
+	p.pageBreakBefore.merge(src.pageBreakBefore)
+	p.widow.merge(src.widow)
+	p.contextual.merge(src.contextual)
+	if src.shd != "" {
+		p.shd = src.shd
+	}
+	if src.borders != nil {
+		if p.borders == nil {
+			p.borders = map[string]docxBorder{}
+		}
+		for k, v := range src.borders {
+			p.borders[k] = v
+		}
+	}
+	if src.numSet {
+		p.numID = src.numID
+		p.numSet = true
+	}
+	p.ilvl.merge(src.ilvl)
+	if len(src.tabs) > 0 {
+		// tab stops accumulate; a "clear" stop removes an inherited one
+		for _, t := range src.tabs {
+			kept := p.tabs[:0]
+			for _, old := range p.tabs {
+				if absF(old.pos-t.pos) > 1 {
+					kept = append(kept, old)
+				}
+			}
+			p.tabs = kept
+			if t.align != "clear" {
+				p.tabs = append(p.tabs, t)
+			}
+		}
+	}
+	if src.sect != nil {
+		p.sect = src.sect
+	}
+	p.outline.merge(src.outline)
+}
+
+/* ---------------- table properties ---------------- */
+
+type docxTblPr struct {
+	style     string
+	width     optNum
+	widthType string
+	ind       optNum
+	jc        string
+	borders   map[string]docxBorder // top left bottom right insideH insideV
+	cellMar   map[string]optNum     // top left bottom right (twips)
+	layout    string
+}
+
+func parseTblPr(n *xnode) docxTblPr {
+	t := docxTblPr{}
+	if n == nil {
+		return t
+	}
+	if s := n.first("tblStyle"); s != nil {
+		t.style = s.attr("val")
+	}
+	if w := n.first("tblW"); w != nil {
+		t.width = numAttr(w, "w")
+		t.widthType = w.attr("type")
+	}
+	if ind := n.first("tblInd"); ind != nil {
+		t.ind = numAttr(ind, "w")
+	}
+	if jc := n.first("jc"); jc != nil {
+		t.jc = jc.attr("val")
+	}
+	if l := n.first("tblLayout"); l != nil {
+		t.layout = l.attr("type")
+	}
+	t.borders = parseBorderSet(n.first("tblBorders"))
+	t.cellMar = parseMarginSet(n.first("tblCellMar"))
+	return t
+}
+
+func parseBorderSet(n *xnode) map[string]docxBorder {
+	if n == nil {
+		return nil
+	}
+	out := map[string]docxBorder{}
+	for _, side := range []string{"top", "left", "bottom", "right", "insideH", "insideV"} {
+		b, ok := parseBorder(n.first(side))
+		if !ok {
+			// start/end are the bidi-neutral spellings of left/right
+			if side == "left" {
+				b, ok = parseBorder(n.first("start"))
+			} else if side == "right" {
+				b, ok = parseBorder(n.first("end"))
+			}
+		}
+		if ok {
+			out[side] = b
+		}
+	}
+	return out
+}
+
+func parseMarginSet(n *xnode) map[string]optNum {
+	if n == nil {
+		return nil
+	}
+	out := map[string]optNum{}
+	for _, side := range []string{"top", "left", "bottom", "right"} {
+		m := n.first(side)
+		if m == nil {
+			if side == "left" {
+				m = n.first("start")
+			} else if side == "right" {
+				m = n.first("end")
+			}
+		}
+		if v := numAttr(m, "w"); v.set {
+			out[side] = v
+		}
+	}
+	return out
+}
+
+func (t *docxTblPr) merge(src docxTblPr) {
+	if src.style != "" {
+		t.style = src.style
+	}
+	if src.width.set {
+		t.width = src.width
+		t.widthType = src.widthType
+	}
+	t.ind.merge(src.ind)
+	if src.jc != "" {
+		t.jc = src.jc
+	}
+	if src.layout != "" {
+		t.layout = src.layout
+	}
+	if src.borders != nil {
+		if t.borders == nil {
+			t.borders = map[string]docxBorder{}
+		}
+		for k, v := range src.borders {
+			t.borders[k] = v
+		}
+	}
+	if src.cellMar != nil {
+		if t.cellMar == nil {
+			t.cellMar = map[string]optNum{}
+		}
+		for k, v := range src.cellMar {
+			t.cellMar[k] = v
+		}
+	}
+}
+
+/* ---------------- style sheet ---------------- */
+
+type docxStyle struct {
+	id, typ, name, basedOn string
+	pPr                    docxPPr
+	rPr                    docxRPr
+	tbl                    docxTblPr
+}
+
+type docxStyleSheet struct {
+	docP        docxPPr
+	docR        docxRPr
+	styles      map[string]*docxStyle
+	defPara     string
+	defTable    string
+	defChar     string
+	resolvedP   map[string]*docxPPr
+	resolvedR   map[string]*docxRPr
+	resolvedTbl map[string]*docxTblPr
+}
+
+func parseStyleSheet(raw []byte) *docxStyleSheet {
+	ss := &docxStyleSheet{
+		styles:      map[string]*docxStyle{},
+		resolvedP:   map[string]*docxPPr{},
+		resolvedR:   map[string]*docxRPr{},
+		resolvedTbl: map[string]*docxTblPr{},
+	}
+	if raw == nil {
+		return ss
+	}
+	tree, err := parseXMLTree(raw)
+	if err != nil {
+		return ss
+	}
+	if dd := tree.first("docDefaults"); dd != nil {
+		ss.docR = parseRPr(dd.path("rPrDefault", "rPr"))
+		ss.docP = parsePPr(dd.path("pPrDefault", "pPr"))
+	}
+	for _, s := range tree.all("style") {
+		st := &docxStyle{id: s.attr("styleId"), typ: s.attr("type")}
+		if n := s.first("name"); n != nil {
+			st.name = n.attr("val")
+		}
+		if b := s.first("basedOn"); b != nil {
+			st.basedOn = b.attr("val")
+		}
+		st.pPr = parsePPr(s.first("pPr"))
+		st.pPr.style = ""
+		st.rPr = parseRPr(s.first("rPr"))
+		st.tbl = parseTblPr(s.first("tblPr"))
+		ss.styles[st.id] = st
+		if onOff2(s.attr("default")) {
+			switch st.typ {
+			case "paragraph":
+				ss.defPara = st.id
+			case "table":
+				ss.defTable = st.id
+			case "character":
+				ss.defChar = st.id
+			}
+		}
+	}
+	return ss
+}
+
+// chain returns the style and its basedOn ancestors, root first
+func (ss *docxStyleSheet) chain(id string) []*docxStyle {
+	var out []*docxStyle
+	seen := map[string]bool{}
+	for id != "" && !seen[id] {
+		seen[id] = true
+		st, ok := ss.styles[id]
+		if !ok {
+			break
+		}
+		out = append([]*docxStyle{st}, out...)
+		id = st.basedOn
+	}
+	return out
+}
+
+// paraStyle resolves a paragraph style (the default one when id is empty)
+// into the pPr/rPr it contributes on top of docDefaults
+func (ss *docxStyleSheet) paraStyle(id string) (*docxPPr, *docxRPr) {
+	if id == "" || ss.styles[id] == nil {
+		id = ss.defPara
+	}
+	if p, ok := ss.resolvedP[id]; ok {
+		return p, ss.resolvedR[id]
+	}
+	p := ss.docP
+	p.style = ""
+	r := ss.docR
+	for _, st := range ss.chain(id) {
+		p.merge(st.pPr)
+		r.merge(st.rPr)
+	}
+	ss.resolvedP[id] = &p
+	ss.resolvedR[id] = &r
+	return &p, &r
+}
+
+// charStyle returns what a character style chain contributes
+func (ss *docxStyleSheet) charStyle(id string) docxRPr {
+	r := docxRPr{}
+	for _, st := range ss.chain(id) {
+		r.merge(st.rPr)
+	}
+	return r
+}
+
+// tableStyle resolves a table style (TableNormal behind it)
+func (ss *docxStyleSheet) tableStyle(id string) *docxTblPr {
+	key := "tbl:" + id
+	if t, ok := ss.resolvedTbl[key]; ok {
+		return t
+	}
+	t := docxTblPr{}
+	if ss.defTable != "" && ss.defTable != id {
+		for _, st := range ss.chain(ss.defTable) {
+			t.merge(st.tbl)
+		}
+	}
+	for _, st := range ss.chain(id) {
+		t.merge(st.tbl)
+	}
+	ss.resolvedTbl[key] = &t
+	return &t
+}
+
+// headingLevel reports which heading (1-6) a paragraph style is, 0 for none.
+// Title and Subtitle come back as -1 and -2.
+func (ss *docxStyleSheet) headingLevel(id string) int {
+	// walk from the style itself towards its roots: a custom style based
+	// on a heading is that heading
+	chain := ss.chain(id)
+	for i := len(chain) - 1; i >= 0; i-- {
+		if lvl := styleNameLevel(chain[i].id, chain[i].name); lvl != 0 {
+			return lvl
+		}
+	}
+	if len(chain) == 0 {
+		return styleNameLevel(id, "")
+	}
+	return 0
+}
+
+func styleNameLevel(id, name string) int {
+	n := strings.ToLower(id)
+	if name != "" {
+		n = strings.ToLower(name)
+	}
+	compact := strings.ReplaceAll(n, " ", "")
+	switch {
+	case compact == "title":
+		return -1
+	case compact == "subtitle":
+		return -2
+	case strings.HasPrefix(compact, "heading") && len(compact) == 8 && compact[7] >= '1' && compact[7] <= '6':
+		return int(compact[7] - '0')
+	}
+	return 0
+}

+ 1840 - 402
src/mod/office/docx_reader.go

@@ -3,21 +3,63 @@ package office
 /*
 /*
 	docx_reader.go - Parse a Word (.docx) file into a Document.
 	docx_reader.go - Parse a Word (.docx) file into a Document.
 
 
-	Converts the common WordprocessingML subset back to the Docs editor
-	HTML: paragraphs, heading/title styles, alignment, bold/italic/
-	underline/strikethrough, font color/size, hyperlinks, bulleted and
-	numbered lists, tables, embedded images (as data URLs), line breaks
-	and page geometry from the section properties. Headers/footers come
-	back as plain text. Tracked changes, footnotes, text boxes and other
-	advanced features are ignored. Legacy binary .doc is rejected.
+	The goal is that an imported document LOOKS like it did where it came
+	from (Word, Google Docs), so formatting is resolved the way Word
+	resolves it - docDefaults, style chains, numbering definitions, table
+	styles - and written out as explicit inline CSS on the editor HTML
+	(see docx_props.go for the inheritance and docs_layout.js for the half
+	of the layout that can only be computed in the browser).
+
+	The HTML vocabulary this produces is the Docs "rich model", also what
+	the docx writer consumes:
+
+	  blocks   p / h1-h6 (.doc-title, .doc-subtitle) with
+	             style: margin-bottom (spacing after), margin-top on headings
+	                    and padding-top elsewhere (spacing before: Google
+	                    Docs collapses a heading's with the spacing above it
+	                    the way CSS margins collapse, and adds a body
+	                    paragraph's), margin-left/-right (indents),
+	                    text-indent, text-align, font-*, color,
+	                    background-color, white-space:pre-wrap when the text
+	                    relies on repeated spaces
+	             data-ls="1.15"       auto line spacing (multiple of single)
+	             data-lsexact="15pt"  exact line height
+	             data-lsmin="15pt"    at-least line height
+	             data-keep-next / data-keep-lines / data-widow = "1"
+	             data-page-break-before="1"
+	             data-tabs="right:451.28:dot;left:36:none" (pt from the
+	                    text column's left edge)
+	  lists    ol / ul .doc-list with data-fmt, data-lvltext, start,
+	             style padding-left + --doc-hang; nested lists sit directly
+	             inside their parent list (what execCommand("indent") makes)
+	  runs     span style=font-*, color, background-color, text-decoration,
+	             font-variant, text-transform; sup / sub; a (links, with
+	             color/decoration inherited so the runs decide),
+	             span.doc-tab (a real tab character), br,
+	             sup.doc-fnref[data-fn] (footnote reference),
+	             span.doc-field[data-field=PAGE|NUMPAGES]
+	  images   img style=width/height in pt, object-view-box for a crop,
+	             border for a picture outline; img.doc-anchor (display:block,
+	             margin offsets) for anchored top-and-bottom pictures
+	  tables   table.of-table style=width/margin-left, colgroup of pt
+	             widths, td/th with explicit border-*, padding,
+	             vertical-align, background-color, colspan/rowspan
+	  breaks   div.doc-pagebreak
+
+	Headers/footers come back as HTML (headerHtml/footerHtml) plus plain
+	text for older consumers, footnotes as body.footnotes. Legacy binary
+	.doc is rejected.
 */
 */
 
 
 import (
 import (
 	"archive/zip"
 	"archive/zip"
 	"bytes"
 	"bytes"
 	"errors"
 	"errors"
+	"fmt"
 	"io"
 	"io"
 	"path"
 	"path"
+	"regexp"
+	"sort"
 	"strconv"
 	"strconv"
 	"strings"
 	"strings"
 )
 )
@@ -36,7 +78,7 @@ func ParseDocx(data []byte) (*Document, error) {
 	for _, f := range zr.File {
 	for _, f := range zr.File {
 		name := path.Clean(f.Name)
 		name := path.Clean(f.Name)
 		if strings.HasSuffix(name, ".xml") || strings.HasSuffix(name, ".rels") ||
 		if strings.HasSuffix(name, ".xml") || strings.HasSuffix(name, ".rels") ||
-			strings.HasPrefix(name, "word/media/") {
+			strings.HasPrefix(name, "word/media/") || strings.HasPrefix(name, "media/") {
 			rc, err := f.Open()
 			rc, err := f.Open()
 			if err != nil {
 			if err != nil {
 				continue
 				continue
@@ -63,64 +105,42 @@ func ParseDocx(data []byte) (*Document, error) {
 		return nil, errors.New("document has no body")
 		return nil, errors.New("document has no body")
 	}
 	}
 
 
-	rels := parseRels(files["word/_rels/document.xml.rels"])
-	numFmt := parseNumberingFormats(files["word/numbering.xml"])
-
-	cv := &docxConv{files: files, rels: rels, numFmt: numFmt, bodyNode: body}
+	cv := &docxConv{
+		files:    files,
+		ss:       parseStyleSheet(files["word/styles.xml"]),
+		nb:       parseNumbering(files["word/numbering.xml"]),
+		bodyNode: body,
+		fnNumber: map[string]int{},
+	}
+	docPart := cv.part("word/document.xml")
+	// Google Docs writes every rsid as zeros and numbers paragraphs from 1
+	cv.gdocs = bytes.Contains(docXML, []byte(`w:rsidR="00000000"`)) &&
+		bytes.Contains(docXML, []byte(`w14:paraId="00000001"`))
 
 
 	doc := &Document{}
 	doc := &Document{}
 
 
-	// page geometry (parsed first: multi-column affects HTML conversion)
-	if sect := body.first("sectPr"); sect != nil {
-		pc := &PageConf{Size: "A4", Orientation: "portrait"}
-		if sz := sect.first("pgSz"); sz != nil {
-			w, _ := strconv.Atoi(sz.attr("w"))
-			h, _ := strconv.Atoi(sz.attr("h"))
-			if sz.attr("orient") == "landscape" || w > h {
-				pc.Orientation = "landscape"
-				w, h = h, w
-			}
-			best := "A4"
-			bestD := 1 << 30
-			for name, dim := range pageSizesTwips {
-				d := abs(dim[0]-w) + abs(dim[1]-h)
-				if d < bestD {
-					bestD = d
-					best = name
-				}
-			}
-			pc.Size = best
-		}
-		if mar := sect.first("pgMar"); mar != nil {
-			m := &MarginsMM{Top: 25.4, Right: 25.4, Bottom: 25.4, Left: 25.4}
-			if v, err := strconv.Atoi(mar.attr("top")); err == nil {
-				m.Top = round1(twipsToMm(v))
-			}
-			if v, err := strconv.Atoi(mar.attr("right")); err == nil {
-				m.Right = round1(twipsToMm(v))
-			}
-			if v, err := strconv.Atoi(mar.attr("bottom")); err == nil {
-				m.Bottom = round1(twipsToMm(v))
-			}
-			if v, err := strconv.Atoi(mar.attr("left")); err == nil {
-				m.Left = round1(twipsToMm(v))
-			}
-			pc.Margins = m
-		}
-		if sect.first("titlePg") != nil {
-			// "different first page" with no first-page part: the editor
-			// calls that "every page except the first"
+	// the document-wide line spacing every paragraph inherits unless it
+	// says otherwise (Google Docs: 1.15, Word: 1.08 or single)
+	cv.defaultLS = 1
+	if dp, _ := cv.ss.paraStyle(""); dp != nil && dp.line.set && (dp.lineRule == "" || dp.lineRule == "auto") && dp.line.v > 0 {
+		cv.defaultLS = round3(dp.line.v / 240)
+	}
+	doc.LineSpacing = cv.defaultLS
+
+	// page geometry (parsed first: margins place anchored pictures and
+	// multi-column layouts change the HTML conversion)
+	sect := body.first("sectPr")
+	if sect != nil {
+		doc.Page = parseSectPage(sect)
+		cv.marginL = twipsAttr(sect.first("pgMar"), "left", 1440) / 20
+		cv.textW = (twipsAttr(sect.first("pgSz"), "w", 11906) -
+			twipsAttr(sect.first("pgMar"), "left", 1440) -
+			twipsAttr(sect.first("pgMar"), "right", 1440)) / 20
+		if sect.first("titlePg") != nil && onOff(sect.first("titlePg")).v {
 			doc.HFMode = HFModeExceptFirst
 			doc.HFMode = HFModeExceptFirst
 		}
 		}
-		if cols := sect.first("cols"); cols != nil {
-			if n, err := strconv.Atoi(cols.attr("num")); err == nil && n > 1 {
-				pc.Columns = n
-				if sp, err := strconv.Atoi(cols.attr("space")); err == nil && sp > 0 {
-					pc.ColGap = round1(twipsToMm(sp))
-				}
-			}
-		}
-		doc.Page = pc
+	} else {
+		cv.marginL, cv.textW = 72, 451.3
 	}
 	}
 
 
 	// Word writes IEEE-style spanning titles as leading single-column
 	// Word writes IEEE-style spanning titles as leading single-column
@@ -128,116 +148,301 @@ func ParseDocx(data []byte) (*Document, error) {
 	if doc.Page != nil && doc.Page.Columns > 1 {
 	if doc.Page != nil && doc.Page.Columns > 1 {
 		cv.markSpanSections(body)
 		cv.markSpanSections(body)
 	}
 	}
-	doc.HTML = cv.blocksToHTML(body)
+	cv.collectSectionBreaks(body)
+	doc.HTML = cv.blocks(body, docPart, blockCtx{top: true})
+
+	// footnotes, in the order the text references them
+	if len(cv.fnOrder) > 0 {
+		if raw, ok := files["word/footnotes.xml"]; ok {
+			if ft, err := parseXMLTree(raw); err == nil {
+				fnPart := cv.part("word/footnotes.xml")
+				byID := map[string]*xnode{}
+				for _, fn := range ft.all("footnote") {
+					byID[fn.attr("id")] = fn
+				}
+				for _, id := range cv.fnOrder {
+					fn := byID[id]
+					if fn == nil {
+						continue
+					}
+					doc.Footnotes = append(doc.Footnotes, Footnote{
+						ID:   id,
+						HTML: cv.blocks(fn, fnPart, blockCtx{footnote: true}),
+					})
+				}
+			}
+		}
+	}
 
 
-	// header / footer text (first part of each kind)
-	for name, raw := range files {
-		if strings.HasPrefix(name, "word/header") && strings.HasSuffix(name, ".xml") && doc.Header == "" {
-			doc.Header = partPlainText(raw)
+	// header / footer: the section's default parts, plus "different first
+	// page" when it is on
+	if sect != nil {
+		hdr := cv.hfPart(sect, "headerReference", "default")
+		ftr := cv.hfPart(sect, "footerReference", "default")
+		if hdr != "" {
+			doc.HeaderHTML, doc.Header = cv.hfHTML(hdr)
 		}
 		}
-		if strings.HasPrefix(name, "word/footer") && strings.HasSuffix(name, ".xml") && doc.Footer == "" {
-			txt, hasPage := footerTextAndPageField(raw)
-			doc.Footer = txt
-			doc.PageNumbers = doc.PageNumbers || hasPage
+		if ftr != "" {
+			doc.FooterHTML, doc.Footer = cv.hfHTML(ftr)
+			doc.PageNumbers = doc.PageNumbers || strings.Contains(doc.FooterHTML, `data-field="PAGE"`)
+		}
+		doc.PageNumbers = doc.PageNumbers || cv.autoPageNumber
+		if doc.HFMode == HFModeExceptFirst {
+			// a first-page part with real content is a different header,
+			// not a missing one; the editor only models "blank on page one"
+			if fh := cv.hfPart(sect, "headerReference", "first"); fh != "" {
+				if h, txt := cv.hfHTML(fh); txt != "" || strings.Contains(h, "<img") {
+					doc.HFMode = ""
+				}
+			}
 		}
 		}
 	}
 	}
 	return doc, nil
 	return doc, nil
 }
 }
 
 
+func round1(v float64) float64 {
+	return float64(int(v*10+0.5)) / 10
+}
+
+func round3(v float64) float64 {
+	if v < 0 {
+		return -round3(-v)
+	}
+	return float64(int(v*1000+0.5)) / 1000
+}
+
 func abs(v int) int {
 func abs(v int) int {
 	if v < 0 {
 	if v < 0 {
 		return -v
 		return -v
 	}
 	}
 	return v
 	return v
 }
 }
-func round1(v float64) float64 {
-	return float64(int(v*10+0.5)) / 10
+
+func twipsAttr(n *xnode, name string, def float64) float64 {
+	if v := numAttr(n, name); v.set {
+		return v.v
+	}
+	return def
 }
 }
 
 
-func partPlainText(raw []byte) string {
-	tree, err := parseXMLTree(raw)
-	if err != nil {
-		return ""
+// parseSectPage reads page size, orientation, margins, header/footer
+// distances and columns from a sectPr
+func parseSectPage(sect *xnode) *PageConf {
+	pc := &PageConf{Size: "A4", Orientation: "portrait"}
+	if sz := sect.first("pgSz"); sz != nil {
+		w := int(twipsAttr(sz, "w", 11906))
+		h := int(twipsAttr(sz, "h", 16838))
+		if sz.attr("orient") == "landscape" || w > h {
+			pc.Orientation = "landscape"
+			w, h = h, w
+		}
+		best := "A4"
+		bestD := 1 << 30
+		for name, dim := range pageSizesTwips {
+			d := abs(dim[0]-w) + abs(dim[1]-h)
+			if d < bestD {
+				bestD = d
+				best = name
+			}
+		}
+		pc.Size = best
 	}
 	}
-	var texts []string
-	collectText(tree, &texts)
-	return strings.TrimSpace(strings.Join(texts, " "))
+	if mar := sect.first("pgMar"); mar != nil {
+		pc.Margins = &MarginsMM{
+			Top:    round1(twipsToMmF(twipsAttr(mar, "top", 1440))),
+			Right:  round1(twipsToMmF(twipsAttr(mar, "right", 1440))),
+			Bottom: round1(twipsToMmF(twipsAttr(mar, "bottom", 1440))),
+			Left:   round1(twipsToMmF(twipsAttr(mar, "left", 1440))),
+		}
+		// Word's top/bottom margin may be negative ("do not move the text
+		// for the header") - the editor has no such notion
+		if pc.Margins.Top < 0 {
+			pc.Margins.Top = -pc.Margins.Top
+		}
+		if pc.Margins.Bottom < 0 {
+			pc.Margins.Bottom = -pc.Margins.Bottom
+		}
+		if v := numAttr(mar, "header"); v.set {
+			d := round1(twipsToMmF(v.v))
+			pc.HeaderDist = &d
+		}
+		if v := numAttr(mar, "footer"); v.set {
+			d := round1(twipsToMmF(v.v))
+			pc.FooterDist = &d
+		}
+	}
+	if cols := sect.first("cols"); cols != nil {
+		if n, err := strconv.Atoi(cols.attr("num")); err == nil && n > 1 {
+			pc.Columns = n
+			if sp := numAttr(cols, "space"); sp.set && sp.v > 0 {
+				pc.ColGap = round1(twipsToMmF(sp.v))
+			}
+		}
+	}
+	return pc
 }
 }
 
 
-// footerTextAndPageField extracts footer text and whether it has a PAGE field
-func footerTextAndPageField(raw []byte) (string, bool) {
-	tree, err := parseXMLTree(raw)
-	if err != nil {
-		return "", false
+func twipsToMmF(tw float64) float64 { return tw * 25.4 / 1440.0 }
+
+/* ---------------- conversion state ---------------- */
+
+type docxPartCtx struct {
+	name string
+	rels map[string]string
+	ext  map[string]bool // rId -> TargetMode External
+}
+
+type blockCtx struct {
+	top      bool // direct children of w:body
+	cell     bool
+	hf       bool
+	footnote bool
+}
+
+type fieldFrame struct {
+	instr    string
+	inResult bool
+}
+
+type docxConv struct {
+	files     map[string][]byte
+	ss        *docxStyleSheet
+	nb        *docxNumDefs
+	bodyNode  *xnode
+	spanIdx   map[int]bool // top-level block indexes that span all columns
+	skipIdx   map[int]bool // empty section-divider paragraphs to drop
+	breakIdx  map[int]bool // paragraphs ending a section that starts a new page
+	defaultLS float64
+	marginL   float64 // pt
+	textW     float64 // pt
+	fields    []fieldFrame
+	fnNumber  map[string]int
+	fnOrder   []string
+	parts     map[string]*docxPartCtx
+	// Google Docs exports: a handful of its layout habits are imitated so
+	// the import matches its own PDF (see the gdocs uses)
+	gdocs bool
+	// spacing-before the next paragraph gives back (see paragraph)
+	reduceBefore float64
+	// a footer carried the page number our export adds (see hfHTML)
+	autoPageNumber bool
+}
+
+// editorTableStyle marks a table BuildDocx wrote from an editor-made one
+const editorTableStyle = "ArozEditorTable"
+
+// autoRuleStyle marks the paragraph BuildDocx writes an <hr> as
+const autoRuleStyle = "ArozHorizontalRule"
+
+// isAutoRule reports a horizontal rule written by BuildDocx: its style and
+// no text
+func isAutoRule(p *xnode) bool {
+	pPr := p.first("pPr")
+	if pPr == nil {
+		return false
+	}
+	ps := pPr.first("pStyle")
+	if ps == nil || ps.attr("val") != autoRuleStyle {
+		return false
 	}
 	}
-	hasPage := false
-	var walk func(n *xnode)
 	var texts []string
 	var texts []string
-	walk = func(n *xnode) {
-		if n.XMLName.Local == "instrText" {
-			if strings.Contains(strings.ToUpper(n.Text), "PAGE") {
-				hasPage = true
+	collectText(p, &texts)
+	return strings.TrimSpace(strings.Join(texts, "")) == ""
+}
+
+// autoPageNumberStyle marks the page-number paragraph BuildDocx adds to a
+// footer, so an import turns it back into Document.PageNumbers
+const autoPageNumberStyle = "ArozPageNumber"
+
+// part loads the relationships of one XML part
+func (cv *docxConv) part(name string) *docxPartCtx {
+	if cv.parts == nil {
+		cv.parts = map[string]*docxPartCtx{}
+	}
+	if p, ok := cv.parts[name]; ok {
+		return p
+	}
+	dir, file := path.Split(name)
+	relsName := dir + "_rels/" + file + ".rels"
+	p := &docxPartCtx{name: name, rels: map[string]string{}, ext: map[string]bool{}}
+	if raw, ok := cv.files[relsName]; ok {
+		if tree, err := parseXMLTree(raw); err == nil {
+			for _, r := range tree.all("Relationship") {
+				p.rels[r.attr("Id")] = r.attr("Target")
+				if strings.EqualFold(r.attr("TargetMode"), "External") {
+					p.ext[r.attr("Id")] = true
+				}
 			}
 			}
-			return
 		}
 		}
-		if n.XMLName.Local == "t" {
-			texts = append(texts, n.Text)
-			return
+	}
+	cv.parts[name] = p
+	return p
+}
+
+// hfPart finds the part path a header/footer reference points at
+func (cv *docxConv) hfPart(sect *xnode, kind, typ string) string {
+	docPart := cv.part("word/document.xml")
+	for _, ref := range sect.all(kind) {
+		t := ref.attr("type")
+		if t == "" {
+			t = "default"
+		}
+		if t != typ {
+			continue
 		}
 		}
-		for i := range n.Nodes {
-			walk(&n.Nodes[i])
+		target := docPart.rels[ref.attrNS("relationships", "id")]
+		if target == "" {
+			continue
 		}
 		}
+		return resolvePartPath("word", target)
 	}
 	}
-	walk(tree)
-	txt := strings.TrimSpace(strings.Join(texts, ""))
-	txt = strings.TrimSuffix(txt, "-")
-	return strings.TrimSpace(txt), hasPage
+	return ""
 }
 }
 
 
-// parseNumberingFormats maps numId -> "bullet"|"decimal" (level 0 format)
-func parseNumberingFormats(raw []byte) map[string]string {
-	out := map[string]string{}
-	if raw == nil {
-		return out
+// hfHTML converts a header/footer part; returns its HTML and plain text
+func (cv *docxConv) hfHTML(partName string) (string, string) {
+	raw, ok := cv.files[partName]
+	if !ok {
+		return "", ""
 	}
 	}
 	tree, err := parseXMLTree(raw)
 	tree, err := parseXMLTree(raw)
 	if err != nil {
 	if err != nil {
-		return out
-	}
-	abstract := map[string]string{} // abstractNumId -> fmt
-	for _, an := range tree.all("abstractNum") {
-		id := an.attr("abstractNumId")
-		if lvl := an.first("lvl"); lvl != nil {
-			if nf := lvl.first("numFmt"); nf != nil {
-				if nf.attr("val") == "bullet" {
-					abstract[id] = "bullet"
-				} else {
-					abstract[id] = "decimal"
+		return "", ""
+	}
+	// the page number our own export adds is the editor's page-number
+	// switch, not footer content
+	kept := tree.Nodes[:0]
+	for _, c := range tree.Nodes {
+		if c.XMLName.Local == "p" {
+			if pPr := c.first("pPr"); pPr != nil {
+				if ps := pPr.first("pStyle"); ps != nil && ps.attr("val") == autoPageNumberStyle {
+					cv.autoPageNumber = true
+					continue
 				}
 				}
 			}
 			}
 		}
 		}
+		kept = append(kept, c)
 	}
 	}
-	for _, num := range tree.all("num") {
-		id := num.attr("numId")
-		if ref := num.first("abstractNumId"); ref != nil {
-			if f, ok := abstract[ref.attr("val")]; ok {
-				out[id] = f
-			}
-		}
+	tree.Nodes = kept
+	saved := cv.fields
+	cv.fields = nil
+	htmlOut := cv.blocks(tree, cv.part(partName), blockCtx{hf: true})
+	cv.fields = saved
+	var texts []string
+	collectText(tree, &texts)
+	txt := strings.TrimSpace(strings.Join(texts, ""))
+	// a header that is only empty paragraphs is no header
+	if txt == "" && !strings.Contains(htmlOut, "<img") {
+		return "", ""
 	}
 	}
-	return out
+	// the editor's plain header, as BuildDocx writes it, stays plain
+	if m := plainHFRe.FindStringSubmatch(htmlOut); m != nil && !strings.Contains(m[1], "<") {
+		return "", txt
+	}
+	return htmlOut, txt
 }
 }
 
 
-/* ---------- conversion ---------- */
-
-type docxConv struct {
-	files    map[string][]byte
-	rels     map[string]string
-	numFmt   map[string]string
-	bodyNode *xnode
-	spanIdx  map[int]bool // top-level block indexes that span all columns
-	skipIdx  map[int]bool // empty section-divider paragraphs to drop
-}
+var plainHFRe = regexp.MustCompile(`^<p><span style="font-size:9pt;color:#6b7078;">(.*)</span></p>$`)
 
 
 // markSpanSections finds paragraph-embedded sectPr elements (section
 // markSpanSections finds paragraph-embedded sectPr elements (section
 // dividers). Blocks belonging to a single-column section of a multi-column
 // dividers). Blocks belonging to a single-column section of a multi-column
@@ -267,7 +472,7 @@ func (cv *docxConv) markSpanSections(body *xnode) {
 						}
 						}
 						cv.spanIdx[i] = true
 						cv.spanIdx[i] = true
 					}
 					}
-					if strings.TrimSpace(cv.runsToHTML(n)) == "" {
+					if paragraphIsEmpty(n) {
 						cv.skipIdx[i] = true // pure divider paragraph
 						cv.skipIdx[i] = true // pure divider paragraph
 					}
 					}
 					pending = nil
 					pending = nil
@@ -279,328 +484,1561 @@ func (cv *docxConv) markSpanSections(body *xnode) {
 	}
 	}
 }
 }
 
 
-// blocksToHTML renders the children of w:body (or a table cell)
-func (cv *docxConv) blocksToHTML(parent *xnode) string {
-	var sb strings.Builder
-	listOpen := "" // "" | "ul" | "ol"
-	closeList := func() {
-		if listOpen != "" {
-			sb.WriteString("</" + listOpen + ">")
-			listOpen = ""
-		}
+// collectSectionBreaks marks the paragraphs that close a section whose
+// successor starts on a new page (single-column documents only - the
+// multi-column mapping above owns those)
+func (cv *docxConv) collectSectionBreaks(body *xnode) {
+	if cv.spanIdx != nil {
+		return
 	}
 	}
-	isTop := parent == cv.bodyNode
-	for i := range parent.Nodes {
-		n := &parent.Nodes[i]
-		if isTop && cv.skipIdx != nil && cv.skipIdx[i] {
-			continue
-		}
-		spanAll := isTop && cv.spanIdx != nil && cv.spanIdx[i]
-		switch n.XMLName.Local {
-		case "p":
-			listKind := "" // "ul" | "ol"
-			if pPr := n.first("pPr"); pPr != nil {
-				if numPr := pPr.first("numPr"); numPr != nil {
-					if nid := numPr.first("numId"); nid != nil {
-						if cv.numFmt[nid.attr("val")] == "decimal" {
-							listKind = "ol"
-						} else {
-							listKind = "ul"
-						}
-					}
-				}
-				// style-based lists (e.g. python-docx "List Bullet")
-				if listKind == "" {
-					if ps := pPr.first("pStyle"); ps != nil {
-						v := ps.attr("val")
-						if strings.HasPrefix(v, "ListBullet") {
-							listKind = "ul"
-						} else if strings.HasPrefix(v, "ListNumber") {
-							listKind = "ol"
-						}
-					}
-				}
-			}
-			if listKind != "" {
-				if listOpen != listKind {
-					closeList()
-					sb.WriteString("<" + listKind + ">")
-					listOpen = listKind
-				}
-				sb.WriteString("<li>" + cv.runsToHTML(n) + "</li>")
-				continue
+	type sp struct {
+		idx int
+		typ string
+	}
+	var list []sp
+	for i := range body.Nodes {
+		n := &body.Nodes[i]
+		if n.XMLName.Local == "p" {
+			if s := n.path("pPr", "sectPr"); s != nil {
+				list = append(list, sp{i, sectType(s)})
 			}
 			}
-			closeList()
-			sb.WriteString(cv.paragraphToHTML(n, spanAll))
-		case "tbl":
-			closeList()
-			sb.WriteString(cv.tableToHTML(n))
 		}
 		}
 	}
 	}
-	closeList()
-	return sb.String()
-}
-
-func (cv *docxConv) paragraphToHTML(p *xnode, spanAll bool) string {
-	tag := "p"
-	var classes []string
-	if spanAll {
-		classes = append(classes, "col-span-all")
+	if len(list) == 0 {
+		return
 	}
 	}
-	align := ""
-	if pPr := p.first("pPr"); pPr != nil {
-		if ps := pPr.first("pStyle"); ps != nil {
-			v := ps.attr("val")
-			switch {
-			case strings.HasPrefix(v, "Heading") && len(v) == 8 && v[7] >= '1' && v[7] <= '6':
-				tag = "h" + string(v[7])
-			case v == "Title":
-				tag = "h1"
-				classes = append(classes, "doc-title")
-			}
-		}
-		if jc := pPr.first("jc"); jc != nil {
-			switch jc.attr("val") {
-			case "center":
-				align = "center"
-			case "right", "end":
-				align = "right"
-			case "both":
-				align = "justify"
-			}
+	cv.breakIdx = map[int]bool{}
+	for k, s := range list {
+		nextType := "nextPage"
+		if k+1 < len(list) {
+			nextType = list[k+1].typ
+		} else if bs := body.first("sectPr"); bs != nil {
+			nextType = sectType(bs)
 		}
 		}
-		if ind := pPr.first("ind"); ind != nil && tag == "p" {
-			if l, err := strconv.Atoi(ind.attr("left")); err == nil && l >= 600 {
-				tag = "blockquote"
-			}
+		if nextType != "continuous" {
+			cv.breakIdx[s.idx] = true
 		}
 		}
 	}
 	}
-	cls := ""
-	if len(classes) > 0 {
-		cls = ` class="` + strings.Join(classes, " ") + `"`
-	}
-	style := ""
-	if align != "" {
-		style = ` style="text-align:` + align + `;"`
+}
+
+func sectType(s *xnode) string {
+	if t := s.first("type"); t != nil && t.attr("val") != "" {
+		return t.attr("val")
 	}
 	}
-	inner := cv.runsToHTML(p)
-	if inner == "" {
-		inner = "<br>"
+	return "nextPage"
+}
+
+func paragraphIsEmpty(p *xnode) bool {
+	var texts []string
+	collectText(p, &texts)
+	if strings.TrimSpace(strings.Join(texts, "")) != "" {
+		return false
 	}
 	}
-	return "<" + tag + cls + style + ">" + inner + "</" + tag + ">"
+	var d []*xnode
+	p.findAll("drawing", &d)
+	return len(d) == 0
+}
+
+/* ---------------- blocks ---------------- */
+
+type listFrame struct {
+	tag   string // ol | ul
+	numID string
+	ilvl  int
+	indL  float64 // pt from the container's left edge
 }
 }
 
 
-// runsToHTML renders the runs (and hyperlinks) of a paragraph
-func (cv *docxConv) runsToHTML(p *xnode) string {
+// blocks renders the block children of w:body, a table cell, a header,
+// footer or footnote
+func (cv *docxConv) blocks(parent *xnode, part *docxPartCtx, ctx blockCtx) string {
 	var sb strings.Builder
 	var sb strings.Builder
-	for i := range p.Nodes {
-		n := &p.Nodes[i]
-		switch n.XMLName.Local {
-		case "r":
-			sb.WriteString(cv.runToHTML(n))
-		case "hyperlink":
-			href := ""
-			for _, a := range n.Attrs {
-				if a.Name.Local == "id" {
-					href = cv.rels[a.Value]
+	var lists []listFrame
+	lastPara := -1 // where the last plain paragraph starts in sb
+	closeLists := func(depth int) {
+		for len(lists) > depth {
+			sb.WriteString("</" + lists[len(lists)-1].tag + ">")
+			lists = lists[:len(lists)-1]
+		}
+	}
+	var walk func(children []xnode, top bool)
+	walk = func(children []xnode, top bool) {
+		for i := range children {
+			n := &children[i]
+			if top && ctx.top {
+				if cv.skipIdx != nil && cv.skipIdx[i] {
+					continue
 				}
 				}
 			}
 			}
-			var inner strings.Builder
-			for j := range n.Nodes {
-				if n.Nodes[j].XMLName.Local == "r" {
-					inner.WriteString(cv.runToHTML(&n.Nodes[j]))
+			switch n.XMLName.Local {
+			case "p":
+				if isAutoRule(n) {
+					closeLists(0)
+					lastPara = -1
+					sb.WriteString("<hr>")
+					continue
+				}
+				spanAll := top && ctx.top && cv.spanIdx != nil && cv.spanIdx[i]
+				item := cv.paragraph(n, part, ctx, spanAll)
+				if item.list != nil {
+					cv.placeListItem(&sb, &lists, item)
+				} else {
+					closeLists(0)
+					lastPara = sb.Len()
+					sb.WriteString(item.html)
+				}
+				if top && ctx.top && cv.breakIdx != nil && cv.breakIdx[i] {
+					closeLists(0)
+					sb.WriteString(pageBreakDiv)
+				}
+			case "tbl":
+				closeLists(0)
+				spanAll := top && ctx.top && cv.spanIdx != nil && cv.spanIdx[i]
+				cv.reduceBefore = 0
+				// Google Docs gives no spacing-after to an empty paragraph
+				// right above a table
+				if cv.gdocs && lastPara >= 0 && emptyBlockHTML(sb.String()[lastPara:]) {
+					cur := sb.String()
+					tail := dropMarginBottom(cur[lastPara:])
+					if tail != cur[lastPara:] {
+						sb.Reset()
+						sb.WriteString(cur[:lastPara])
+						sb.WriteString(tail)
+					}
+				}
+				sb.WriteString(cv.table(n, part, spanAll))
+			case "sdt":
+				if c := n.first("sdtContent"); c != nil {
+					walk(c.Nodes, false)
+				}
+			case "customXml", "ins", "moveTo", "smartTag":
+				walk(n.Nodes, false)
+			case "AlternateContent":
+				if c := n.first("Choice"); c != nil {
+					walk(c.Nodes, false)
+				} else if f := n.first("Fallback"); f != nil {
+					walk(f.Nodes, false)
 				}
 				}
-			}
-			if href != "" {
-				sb.WriteString(`<a href="` + xmlEscape(href) + `">` + inner.String() + "</a>")
-			} else {
-				sb.WriteString(inner.String())
 			}
 			}
 		}
 		}
 	}
 	}
+	walk(parent.Nodes, true)
+	closeLists(0)
 	return sb.String()
 	return sb.String()
 }
 }
 
 
-func (cv *docxConv) runToHTML(r *xnode) string {
-	var open, close string
-	var styleProps []string
-	if rPr := r.first("rPr"); rPr != nil {
-		if rPr.first("b") != nil && rPr.first("b").attr("val") != "0" && rPr.first("b").attr("val") != "false" {
-			open += "<b>"
-			close = "</b>" + close
+var marginBottomRe = regexp.MustCompile(`margin-bottom:[^;"]*;`)
+
+var emptyBlockRe = regexp.MustCompile(`^<[a-z0-9]+[^>]*><br></[a-z0-9]+>$`)
+
+// emptyBlockHTML reports whether s is one block holding nothing but its
+// empty line
+func emptyBlockHTML(s string) bool {
+	return emptyBlockRe.MatchString(s)
+}
+
+// dropMarginBottom removes the spacing-after from the opening tag of the
+// block HTML that starts s (only when s is a single block)
+func dropMarginBottom(s string) string {
+	end := strings.Index(s, ">")
+	if end < 0 {
+		return s
+	}
+	return marginBottomRe.ReplaceAllString(s[:end], "margin-bottom:0pt;") + s[end:]
+}
+
+const pageBreakDiv = `<div class="doc-pagebreak" contenteditable="false"></div>`
+
+type paraOutput struct {
+	html    string // complete block HTML (non-list paragraphs)
+	list    *docxListItem
+	liAttrs string
+	liInner string
+}
+
+type docxListItem struct {
+	numID   string
+	ilvl    int
+	fmt     string
+	text    string
+	start   int
+	value   int
+	indL    float64 // pt
+	hang    float64 // pt
+	ordered bool
+}
+
+// placeListItem opens/closes list elements around one list paragraph
+func (cv *docxConv) placeListItem(sb *strings.Builder, lists *[]listFrame, it paraOutput) {
+	li := it.list
+	// a different list instance closes the current one entirely
+	if len(*lists) > 0 && (*lists)[0].numID != li.numID {
+		for len(*lists) > 0 {
+			sb.WriteString("</" + (*lists)[len(*lists)-1].tag + ">")
+			*lists = (*lists)[:len(*lists)-1]
 		}
 		}
-		if rPr.first("i") != nil {
-			open += "<i>"
-			close = "</i>" + close
+	}
+	for len(*lists) > 0 && (*lists)[len(*lists)-1].ilvl > li.ilvl {
+		sb.WriteString("</" + (*lists)[len(*lists)-1].tag + ">")
+		*lists = (*lists)[:len(*lists)-1]
+	}
+	tag := "ul"
+	if li.ordered {
+		tag = "ol"
+	}
+	if n := len(*lists); n > 0 && (*lists)[n-1].ilvl == li.ilvl && (*lists)[n-1].tag != tag {
+		sb.WriteString("</" + (*lists)[n-1].tag + ">")
+		*lists = (*lists)[:n-1]
+	}
+	if len(*lists) == 0 || (*lists)[len(*lists)-1].ilvl < li.ilvl {
+		parentL := 0.0
+		if len(*lists) > 0 {
+			parentL = (*lists)[len(*lists)-1].indL
 		}
 		}
-		if u := rPr.first("u"); u != nil && u.attr("val") != "none" {
-			open += "<u>"
-			close = "</u>" + close
+		pad := li.indL - parentL
+		attrs := ` class="doc-list" data-num="` + xmlEscape(li.numID) + `" data-fmt="` + li.fmt + `"`
+		if li.text != "" {
+			attrs += ` data-lvltext="` + xmlEscape(li.text) + `"`
 		}
 		}
-		if rPr.first("strike") != nil {
-			open += "<s>"
-			close = "</s>" + close
+		if li.ordered && li.value != 1 {
+			attrs += ` start="` + strconv.Itoa(li.value) + `"`
 		}
 		}
-		if c := rPr.first("color"); c != nil {
-			v := c.attr("val")
-			if len(v) == 6 && v != "000000" && strings.ToUpper(v) != "AUTO" {
-				styleProps = append(styleProps, "color:#"+strings.ToLower(v))
-			}
+		attrs += ` style="padding-left:` + ptStr(pad) + `;--doc-hang:` + ptStr(li.hang) + `;"`
+		sb.WriteString("<" + tag + attrs + ">")
+		*lists = append(*lists, listFrame{tag: tag, numID: li.numID, ilvl: li.ilvl, indL: li.indL})
+	}
+	sb.WriteString("<li" + it.liAttrs + ">" + it.liInner + "</li>")
+}
+
+// ptStr formats a length in points
+func ptStr(v float64) string {
+	return trimFloat(round2(v)) + "pt"
+}
+
+/* ---------------- paragraphs ---------------- */
+
+// paragraph converts one w:p
+func (cv *docxConv) paragraph(p *xnode, part *docxPartCtx, ctx blockCtx, spanAll bool) paraOutput {
+	direct := parsePPr(p.first("pPr"))
+	styleP, styleR := cv.ss.paraStyle(direct.style)
+	eff := *styleP
+	eff.merge(direct)
+	baseR := *styleR
+	baseR.merge(direct.mark)
+	if cv.reduceBefore > 0 && !ctx.cell {
+		b := 0.0
+		if eff.before.set {
+			b = eff.before.v - cv.reduceBefore*20
 		}
 		}
-		if sz := rPr.first("sz"); sz != nil {
-			if hp, err := strconv.ParseFloat(sz.attr("val"), 64); err == nil && hp > 0 && hp != 22 {
-				px := halfPointsToPx(hp)
-				styleProps = append(styleProps, "font-size:"+strconv.Itoa(int(px+0.5))+"px")
-			}
+		if b < 0 {
+			b = 0
 		}
 		}
+		eff.before = optNum{set: true, v: b}
 	}
 	}
-	var body strings.Builder
-	for i := range r.Nodes {
-		n := &r.Nodes[i]
-		switch n.XMLName.Local {
-		case "t":
-			body.WriteString(xmlEscape(n.Text))
-		case "br", "cr":
-			if n.attr("type") == "page" {
-				// explicit page break -> the editor's page break block
-				body.WriteString(`<div class="doc-pagebreak" contenteditable="false"></div>`)
-			} else {
-				body.WriteString("<br>")
+	cv.reduceBefore = 0
+
+	level := cv.ss.headingLevel(direct.style)
+	if direct.style == "" {
+		level = cv.ss.headingLevel(cv.ss.defPara)
+	}
+
+	// list membership (numbering from the paragraph or its style)
+	var li *docxListItem
+	if eff.numSet && cv.nb.exists(eff.numID) && !ctx.hf {
+		ilvl := 0
+		if eff.ilvl.set {
+			ilvl = int(eff.ilvl.v)
+		}
+		def := cv.nb.level(eff.numID, ilvl)
+		li = &docxListItem{numID: eff.numID, ilvl: ilvl, fmt: "bullet", start: 1}
+		if def != nil {
+			li.fmt = htmlListFormat(def.fmt)
+			li.text = def.text
+			li.start = def.start
+			// numbering indents sit between the style's and the paragraph's own
+			ind := *styleP
+			if def.indL.set {
+				ind.indL = def.indL
 			}
 			}
-		case "tab":
-			body.WriteString("&nbsp;&nbsp;&nbsp;&nbsp;")
-		case "drawing", "pict", "object":
-			body.WriteString(cv.imageToHTML(n))
+			if def.indHang.set || def.indFirst.set {
+				ind.indHang, ind.indFirst = def.indHang, def.indFirst
+			}
+			ind.merge(direct)
+			eff.indL, eff.indHang, eff.indFirst = ind.indL, ind.indHang, ind.indFirst
+		}
+		li.ordered = li.fmt != "bullet"
+		li.value = cv.nb.next(eff.numID, ilvl)
+		li.indL = eff.indL.v / 20
+		if eff.indHang.set && eff.indHang.v > 0 {
+			li.hang = eff.indHang.v / 20
+		} else if eff.indFirst.set && eff.indFirst.v < 0 {
+			li.hang = -eff.indFirst.v / 20
+		} else {
+			li.hang = 18
 		}
 		}
 	}
 	}
-	out := body.String()
-	if out == "" {
-		return ""
+
+	runs := cv.runs(p, part, baseR)
+
+	tag := "p"
+	var classes []string
+	switch {
+	case level >= 1 && level <= 6 && li == nil:
+		tag = "h" + strconv.Itoa(level)
+	case level == -1 && li == nil:
+		tag = "h1"
+		classes = append(classes, "doc-title")
+	case level == -2 && li == nil:
+		classes = append(classes, "doc-subtitle")
 	}
 	}
-	if len(styleProps) > 0 {
-		open += `<span style="` + strings.Join(styleProps, ";") + `;">`
-		close = "</span>" + close
+	if spanAll {
+		classes = append(classes, "col-span-all")
 	}
 	}
-	return open + out + close
-}
 
 
-// imageToHTML finds the blip relationship inside a drawing and inlines it
-func (cv *docxConv) imageToHTML(n *xnode) string {
-	rid := ""
-	var wPx int
-	var findBlip func(x *xnode)
-	findBlip = func(x *xnode) {
-		if x.XMLName.Local == "blip" {
-			for _, a := range x.Attrs {
-				if a.Name.Local == "embed" {
-					rid = a.Value
-				}
-			}
+	style, data := cv.blockStyle(tag, eff, baseR, li != nil, level, runs.preWrap)
+	attrs := ""
+	if runs.bookmark != "" {
+		attrs += ` id="` + xmlEscape(runs.bookmark) + `"`
+	}
+	if len(classes) > 0 {
+		attrs += ` class="` + strings.Join(classes, " ") + `"`
+	}
+	if style != "" {
+		attrs += ` style="` + style + `"`
+	}
+	attrs += data
+
+	if li != nil {
+		inner := strings.Join(runs.segments, "")
+		if strings.TrimSpace(stripTags(inner)) == "" && !strings.Contains(inner, "<img") {
+			inner += "<br>"
+		}
+		return paraOutput{list: li, liAttrs: attrs, liInner: inner}
+	}
+
+	var sb strings.Builder
+	for i, seg := range runs.segments {
+		if i > 0 {
+			sb.WriteString(pageBreakDiv)
+		}
+		// an anchored picture sits above/below the text, so a paragraph
+		// holding only that still has its own (empty) line
+		flow := anchorImgRe.ReplaceAllString(seg, "")
+		empty := strings.TrimSpace(stripTags(flow)) == "" && !strings.Contains(flow, "<img") &&
+			!strings.Contains(flow, "doc-tab")
+		anchorsOnly := empty && flow != seg
+		if anchorsOnly {
+			seg += "<br>"
+			empty = false
+		}
+		if len(runs.segments) > 1 && empty {
+			// the part of a paragraph before or after a page break that
+			// holds nothing takes no line (how Google Docs lays it out)
+			continue
+		}
+		if empty && !strings.Contains(seg, "<br>") {
+			seg += "<br>"
+		} else if strings.HasSuffix(seg, "<br>") && !anchorsOnly {
+			// a trailing line break opens one more (empty) line
+			seg += "<br>"
 		}
 		}
-		if x.XMLName.Local == "extent" && wPx == 0 {
-			if cx, err := strconv.ParseInt(x.attr("cx"), 10, 64); err == nil {
-				wPx = int(emuToPx(cx, 1.0))
+		segAttrs := attrs
+		if i > 0 {
+			segAttrs = strings.Replace(segAttrs, ` id="`+xmlEscape(runs.bookmark)+`"`, "", 1)
+			// the text after a page break carries on the paragraph that
+			// began on the page before: its spacing-before was spent there
+			noBefore := eff
+			noBefore.before = optNum{set: true, v: 0}
+			st2, _ := cv.blockStyle(tag, noBefore, baseR, false, level, runs.preWrap)
+			if style != "" {
+				segAttrs = strings.Replace(segAttrs, ` style="`+style+`"`, ` style="`+st2+`"`, 1)
 			}
 			}
 		}
 		}
-		for i := range x.Nodes {
-			findBlip(&x.Nodes[i])
+		sb.WriteString("<" + tag + segAttrs + ">" + seg + "</" + tag + ">")
+	}
+	// A paragraph that ends in a page break leaves an empty remainder on
+	// the next page that takes no line - but Google Docs still counts its
+	// spacing-after against the spacing-before of what follows (a heading
+	// after a page-break heading starts 4pt higher than one after a
+	// page-break body paragraph)
+	if n := len(runs.segments); n > 1 && !ctx.cell && eff.after.set {
+		last := runs.segments[n-1]
+		if strings.TrimSpace(stripTags(last)) == "" && !strings.Contains(last, "<img") {
+			cv.reduceBefore = eff.after.v / 20
 		}
 		}
 	}
 	}
-	findBlip(n)
-	if rid == "" {
-		return ""
+	return paraOutput{html: sb.String()}
+}
+
+// the editor's own defaults for body text (docs.css) - anything else is
+// stated inline
+const (
+	editorBodyPt    = 11.0
+	editorBodyColor = "000000"
+)
+
+var editorFontStack = docxFontStack("Arial", "")
+
+var anchorImgRe = regexp.MustCompile(`<img[^>]*class="doc-anchor"[^>]*>`)
+
+// font names that stand for another face in practice: Google Docs writes
+// "Arial Unicode MS" for runs holding symbols and lays them out in Arial,
+// while a Windows machine that has the old Office font installed would draw
+// them 20% taller and wider
+var docxFontAlias = map[string]string{
+	"arial unicode ms": "Arial",
+}
+
+// metric twins a word processor substitutes for a font it does not have
+var docxMetricFallback = map[string]string{
+	"sans": "Arial", "serif": "'Times New Roman'", "mono": "'Courier New'",
+}
+
+// docxFontStack is fontStackFor with the substitution Google Docs and Word
+// make for a font they do not have: a metric-compatible core font right
+// behind it. Without that the browser falls through to the shipped Noto
+// faces, which are wider and taller, and every line wraps and stacks
+// differently from the source.
+func docxFontStack(latin, ea string) string {
+	if alias, ok := docxFontAlias[strings.ToLower(latin)]; ok {
+		latin = alias
 	}
 	}
-	target, ok := cv.rels[rid]
-	if !ok {
-		return ""
+	if alias, ok := docxFontAlias[strings.ToLower(ea)]; ok {
+		ea = alias
 	}
 	}
-	mediaPath := resolvePartPath("word", target)
-	data, ok2 := cv.files[mediaPath]
-	if !ok2 {
-		return ""
+	if ea == latin {
+		ea = ""
 	}
 	}
-	ext := strings.TrimPrefix(strings.ToLower(path.Ext(mediaPath)), ".")
-	attrs := ""
-	if wPx > 10 {
-		attrs = ` style="width:` + strconv.Itoa(wPx) + `px;"`
+	stack := fontStackFor(latin, ea)
+	l := strings.ToLower(latin)
+	kind := "sans"
+	switch {
+	case strings.Contains(l, "courier"), strings.Contains(l, "mono"), strings.Contains(l, "consolas"):
+		kind = "mono"
+	case strings.Contains(l, "times"), strings.Contains(l, "georgia"), strings.Contains(l, "garamond"),
+		strings.Contains(l, "cambria"), strings.Contains(l, "book antiqua"),
+		strings.Contains(l, "serif") && !strings.Contains(l, "sans"):
+		kind = "serif"
 	}
 	}
-	return `<img src="` + encodeDataURL(data, ext) + `"` + attrs + ">"
+	fb := docxMetricFallback[kind]
+	if strings.EqualFold(strings.Trim(fb, "'"), latin) {
+		return stack
+	}
+	first := quoteFontName(latin)
+	if ea != "" && ea != latin {
+		first += "," + quoteFontName(ea)
+	}
+	if strings.HasPrefix(stack, first+",") {
+		return first + "," + fb + stack[len(first):]
+	}
+	return stack
 }
 }
 
 
-func (cv *docxConv) tableToHTML(tbl *xnode) string {
-	var sb strings.Builder
-	// table width: tblW pct (fiftieths of a percent) or dxa (twips of the
-	// ~9026-twip text column); auto/absent = the editor's default 100%
-	widthStyle := ""
-	if tblPr := tbl.first("tblPr"); tblPr != nil {
-		if tw := tblPr.first("tblW"); tw != nil {
-			if v, err := strconv.ParseFloat(tw.attr("w"), 64); err == nil && v > 0 {
-				pct := 0.0
-				switch tw.attr("type") {
-				case "pct":
-					pct = v / 50.0
-				case "dxa":
-					pct = v * 100 / 9026.0
-				}
-				if pct > 100 {
-					pct = 100
-				}
-				// full-width tables need no inline style
-				if pct > 1 && pct < 99.5 {
-					widthStyle = ` style="width:` + trimFloat(pct) + `%;"`
-				}
-			}
+// blockStyle renders a paragraph's layout properties as inline CSS plus
+// data attributes
+func (cv *docxConv) blockStyle(tag string, eff docxPPr, r docxRPr, isList bool, level int, preWrap bool) (string, string) {
+	var css []string
+	add := func(k, v string) { css = append(css, k+":"+v) }
+	heading := tag != "p"
+
+	// a paragraph border sits between the spacing and the text: the text
+	// keeps its indent and the rule is drawn "space" points outside it
+	edge := func(side string) (float64, bool) {
+		b, ok := eff.borders[side]
+		if !ok || !b.visible() {
+			return 0, false
 		}
 		}
+		return borderWidthPt(b) + b.space, true
 	}
 	}
-	sb.WriteString(`<table class="of-table"` + widthStyle + `>`)
-	// column proportions -> the editor's colgroup
-	if grid := tbl.first("tblGrid"); grid != nil {
-		var ws []float64
-		sum := 0.0
-		for _, gc := range grid.all("gridCol") {
-			if v, err := strconv.ParseFloat(gc.attr("w"), 64); err == nil && v > 0 {
-				ws = append(ws, v)
-				sum += v
-			}
+	bordered := false
+	for _, side := range []string{"top", "right", "bottom", "left"} {
+		if _, ok := edge(side); ok {
+			bordered = true
 		}
 		}
-		if len(ws) > 1 && sum > 0 {
-			sb.WriteString("<colgroup>")
-			for _, w := range ws {
-				sb.WriteString(`<col style="width:` + trimFloat(w*100/sum) + `%">`)
-			}
-			sb.WriteString("</colgroup>")
+	}
+	before, after := 0.0, 0.0
+	if eff.before.set {
+		before = eff.before.v / 20
+	}
+	if eff.after.set {
+		after = eff.after.v / 20
+	}
+	if before != 0 || heading {
+		// Google Docs collapses a heading's spacing-before with the
+		// spacing-after above it (the larger wins) but adds a body
+		// paragraph's to it - margin collapses in CSS, padding does not
+		if heading || bordered {
+			// (a bordered paragraph's spacing is outside its rule)
+			add("margin-top", ptStr(before))
+		} else {
+			add("padding-top", ptStr(before))
 		}
 		}
 	}
 	}
-	for i := range tbl.Nodes {
-		tr := &tbl.Nodes[i]
-		if tr.XMLName.Local != "tr" {
-			continue
+	if after != 0 || heading {
+		add("margin-bottom", ptStr(after))
+	}
+	if !isList {
+		left := 0.0
+		if eff.indL.set {
+			left = eff.indL.v / 20
 		}
 		}
-		sb.WriteString("<tr>")
-		for j := range tr.Nodes {
-			tc := &tr.Nodes[j]
-			if tc.XMLName.Local != "tc" {
-				continue
-			}
-			// cell shading survives as an inline background
-			tdStyle := ""
-			if tcPr := tc.first("tcPr"); tcPr != nil {
-				if shd := tcPr.first("shd"); shd != nil {
-					if fill := shd.attr("fill"); len(fill) == 6 && fill != "auto" {
-						tdStyle = ` style="background-color:#` + strings.ToLower(fill) + `;"`
-					}
-				}
-			}
-			inner := cv.blocksToHTML(tc)
-			// unwrap a single plain paragraph for cleaner cells
-			if strings.HasPrefix(inner, "<p>") && strings.HasSuffix(inner, "</p>") &&
-				strings.Count(inner, "<p>") == 1 {
-				inner = strings.TrimSuffix(strings.TrimPrefix(inner, "<p>"), "</p>")
+		if shift, ok := edge("left"); ok {
+			left -= shift
+		}
+		if left != 0 {
+			add("margin-left", ptStr(left))
+		}
+		first := 0.0
+		if eff.indHang.set && eff.indHang.v != 0 {
+			first = -eff.indHang.v / 20
+		} else if eff.indFirst.set {
+			first = eff.indFirst.v / 20
+		}
+		if first != 0 {
+			add("text-indent", ptStr(first))
+		}
+	}
+	right := 0.0
+	if eff.indR.set {
+		right = eff.indR.v / 20
+	}
+	if shift, ok := edge("right"); ok && !isList {
+		right -= shift
+	}
+	if right != 0 {
+		add("margin-right", ptStr(right))
+	}
+	switch eff.jc {
+	case "center":
+		add("text-align", "center")
+	case "right", "end":
+		add("text-align", "right")
+	case "both", "distribute":
+		add("text-align", "justify")
+	}
+
+	// run defaults of the block
+	fc := rPrCSS(r)
+	if fc["font-family"] != editorFontStack || heading {
+		add("font-family", fc["font-family"])
+	}
+	if fc["font-size"] != ptStr(editorBodyPt) || heading {
+		add("font-size", fc["font-size"])
+	}
+	if fc["font-weight"] != "400" || heading {
+		add("font-weight", fc["font-weight"])
+	}
+	if fc["font-style"] != "normal" || heading {
+		add("font-style", fc["font-style"])
+	}
+	if fc["color"] != "#"+strings.ToLower(editorBodyColor) || heading {
+		add("color", fc["color"])
+	}
+	for _, k := range []string{"text-decoration", "font-variant", "text-transform", "background-color"} {
+		if v := fc[k]; v != "" && v != "none" && v != "normal" {
+			add(k, v)
+		}
+	}
+	if eff.shd != "" && eff.shd != "auto" {
+		add("background-color", "#"+strings.ToLower(eff.shd))
+	}
+	for _, side := range []string{"top", "right", "bottom", "left"} {
+		b, ok := eff.borders[side]
+		if !ok || !b.visible() {
+			continue
+		}
+		add("border-"+side, borderCSS(b))
+		if b.space > 0 {
+			add("padding-"+side, ptStr(b.space))
+		}
+	}
+	if preWrap {
+		add("white-space", "pre-wrap")
+	}
+
+	var data strings.Builder
+	switch {
+	case eff.line.set && eff.lineRule == "exact":
+		data.WriteString(` data-lsexact="` + ptStr(eff.line.v/20) + `"`)
+	case eff.line.set && eff.lineRule == "atLeast":
+		data.WriteString(` data-lsmin="` + ptStr(eff.line.v/20) + `"`)
+	case eff.line.set && eff.line.v > 0:
+		if ls := round3(eff.line.v / 240); ls != cv.defaultLS {
+			data.WriteString(` data-ls="` + trimFloat(ls) + `"`)
+		}
+	case cv.defaultLS != 1:
+		data.WriteString(` data-ls="1"`)
+	}
+	if eff.keepNext.v {
+		data.WriteString(` data-keep-next="1"`)
+	}
+	if eff.keepLines.v {
+		data.WriteString(` data-keep-lines="1"`)
+	}
+	// widow/orphan control is on unless a paragraph turns it off (Word's
+	// Normal style and Google Docs both keep two lines together)
+	if eff.widow.set && !eff.widow.v {
+		data.WriteString(` data-widow="0"`)
+	}
+	if eff.pageBreakBefore.v {
+		data.WriteString(` data-page-break-before="1"`)
+	}
+	if len(eff.tabs) > 0 {
+		tabs := append([]docxTab(nil), eff.tabs...)
+		sort.Slice(tabs, func(i, j int) bool { return tabs[i].pos < tabs[j].pos })
+		var parts []string
+		for _, t := range tabs {
+			if t.align == "clear" {
+				continue
+			}
+			align := t.align
+			switch align {
+			case "start", "":
+				align = "left"
+			case "end":
+				align = "right"
+			}
+			leader := t.leader
+			if leader == "" {
+				leader = "none"
+			}
+			parts = append(parts, align+":"+trimFloat(round2(t.pos/20))+":"+leader)
+		}
+		if len(parts) > 0 {
+			data.WriteString(` data-tabs="` + strings.Join(parts, ";") + `"`)
+		}
+	}
+	if len(css) == 0 {
+		return "", data.String()
+	}
+	return xmlEscape(strings.Join(css, ";") + ";"), data.String()
+}
+
+// borderWidthPt is the width a border is drawn at, in points
+func borderWidthPt(b docxBorder) float64 {
+	w := b.sz / 8
+	if w <= 0 {
+		w = 0.5
+	}
+	if b.val == "double" && w < 2.25 {
+		w = 2.25
+	}
+	return w
+}
+
+func borderCSS(b docxBorder) string {
+	w := borderWidthPt(b)
+	style := "solid"
+	switch b.val {
+	case "dotted":
+		style = "dotted"
+	case "dashed", "dashSmallGap", "dotDash", "dotDotDash":
+		style = "dashed"
+	case "double":
+		style = "double"
+	}
+	col := b.color
+	if len(col) != 6 {
+		col = "000000"
+	}
+	return ptStr(w) + " " + style + " #" + strings.ToLower(col)
+}
+
+var highlightColors = map[string]string{
+	"yellow": "ffff00", "green": "00ff00", "cyan": "00ffff", "magenta": "ff00ff",
+	"blue": "0000ff", "red": "ff0000", "darkBlue": "000080", "darkCyan": "008080",
+	"darkGreen": "008000", "darkMagenta": "800080", "darkRed": "800000",
+	"darkYellow": "808000", "darkGray": "808080", "lightGray": "c0c0c0",
+	"black": "000000", "white": "ffffff",
+}
+
+// rPrCSS renders a fully resolved run's formatting as CSS values
+func rPrCSS(r docxRPr) map[string]string {
+	out := map[string]string{}
+	font := r.fontASCII
+	if font == "" {
+		font = "Arial"
+	}
+	ea := r.fontEA
+	if ea == font {
+		ea = ""
+	}
+	out["font-family"] = docxFontStack(font, ea)
+	sz := 20.0 // Word's built-in default is 10pt
+	if r.sz.set && r.sz.v > 0 {
+		sz = r.sz.v
+	}
+	out["font-size"] = ptStr(sz / 2)
+	out["font-weight"] = "400"
+	if r.b.v {
+		out["font-weight"] = "700"
+	}
+	out["font-style"] = "normal"
+	if r.i.v {
+		out["font-style"] = "italic"
+	}
+	var deco []string
+	if r.u != "" && r.u != "none" {
+		deco = append(deco, "underline")
+	}
+	if r.strike.v || r.dstrike.v {
+		deco = append(deco, "line-through")
+	}
+	out["text-decoration"] = "none"
+	if len(deco) > 0 {
+		out["text-decoration"] = strings.Join(deco, " ")
+	}
+	col := r.color
+	if len(col) != 6 {
+		col = editorBodyColor
+	}
+	out["color"] = "#" + strings.ToLower(col)
+	if r.highlight != "" && r.highlight != "none" {
+		if h, ok := highlightColors[r.highlight]; ok {
+			out["background-color"] = "#" + h
+		}
+	} else if r.shd != "" && r.shd != "auto" {
+		out["background-color"] = "#" + strings.ToLower(r.shd)
+	}
+	if r.smallCaps.v {
+		out["font-variant"] = "small-caps"
+	}
+	if r.caps.v {
+		out["text-transform"] = "uppercase"
+	}
+	return out
+}
+
+var cssRunKeys = []string{"font-family", "font-size", "font-weight", "font-style",
+	"text-decoration", "color", "background-color", "font-variant", "text-transform"}
+
+// runCSSDiff renders the properties in which a run differs from its block
+func runCSSDiff(run, block map[string]string) string {
+	var parts []string
+	for _, k := range cssRunKeys {
+		rv, bv := run[k], block[k]
+		if rv == bv {
+			continue
+		}
+		if rv == "" {
+			switch k {
+			case "background-color":
+				rv = "transparent"
+			case "font-variant", "text-transform":
+				rv = "normal"
+				if k == "text-transform" {
+					rv = "none"
+				}
+			default:
+				continue
+			}
+		}
+		parts = append(parts, k+":"+rv)
+	}
+	if len(parts) == 0 {
+		return ""
+	}
+	return strings.Join(parts, ";") + ";"
+}
+
+/* ---------------- runs ---------------- */
+
+type runsResult struct {
+	segments []string // inline HTML, split at page breaks
+	preWrap  bool
+	bookmark string
+	midText  bool // the text so far ends in a non-space, so a leading space is kept
+}
+
+type inlinePiece struct {
+	css  string // run style diff
+	vert string
+	link string
+	html string
+	raw  bool // html is a complete element (no span wrapping)
+}
+
+// runs renders the inline content of a paragraph
+func (cv *docxConv) runs(p *xnode, part *docxPartCtx, baseR docxRPr) runsResult {
+	res := runsResult{}
+	blockCSS := rPrCSS(baseR)
+	var pieces []inlinePiece
+	var segments []string
+	flush := func() {
+		segments = append(segments, joinPieces(pieces))
+		pieces = nil
+	}
+
+	var walk func(n *xnode, link string)
+	walk = func(n *xnode, link string) {
+		for i := range n.Nodes {
+			c := &n.Nodes[i]
+			switch c.XMLName.Local {
+			case "r":
+				cv.run(c, part, baseR, blockCSS, link, &pieces, &res, flush)
+			case "hyperlink":
+				href := ""
+				if id := c.attrNS("relationships", "id"); id != "" {
+					href = part.rels[id]
+				}
+				if a := c.attr("anchor"); a != "" && href == "" {
+					href = "#" + a
+				}
+				walk(c, href)
+			case "fldSimple":
+				instr := strings.ToUpper(strings.TrimSpace(c.attr("instr")))
+				cv.fields = append(cv.fields, fieldFrame{instr: instr, inResult: true})
+				walk(c, link)
+				cv.fields = cv.fields[:len(cv.fields)-1]
+			case "smartTag", "customXml", "ins", "moveTo", "bdo", "dir":
+				walk(c, link)
+			case "sdt":
+				if sc := c.first("sdtContent"); sc != nil {
+					walk(sc, link)
+				}
+			case "AlternateContent":
+				if ch := c.first("Choice"); ch != nil {
+					walk(ch, link)
+				} else if fb := c.first("Fallback"); fb != nil {
+					walk(fb, link)
+				}
+			case "bookmarkStart":
+				name := c.attr("name")
+				if res.bookmark == "" && name != "" && name != "_GoBack" {
+					res.bookmark = name
+				}
+			}
+		}
+	}
+	walk(p, "")
+	flush()
+	res.segments = segments
+	return res
+}
+
+// fieldHidden reports whether we are inside a field's instruction text
+func (cv *docxConv) fieldHidden() bool {
+	for _, f := range cv.fields {
+		if !f.inResult {
+			return true
+		}
+	}
+	return false
+}
+
+// fieldKind names the innermost PAGE / NUMPAGES field being shown
+func (cv *docxConv) fieldKind() string {
+	for i := len(cv.fields) - 1; i >= 0; i-- {
+		w := strings.Fields(cv.fields[i].instr)
+		if len(w) > 0 && (w[0] == "PAGE" || w[0] == "NUMPAGES") {
+			return w[0]
+		}
+	}
+	return ""
+}
+
+func (cv *docxConv) run(r *xnode, part *docxPartCtx, baseR docxRPr, blockCSS map[string]string,
+	link string, pieces *[]inlinePiece, res *runsResult, pageBreak func()) {
+
+	direct := parseRPr(r.first("rPr"))
+	eff := baseR
+	if direct.rStyle != "" {
+		eff.merge(cv.ss.charStyle(direct.rStyle))
+	}
+	eff.merge(direct)
+	if eff.vanish.v {
+		return
+	}
+	css := runCSSDiff(rPrCSS(eff), blockCSS)
+	vert := ""
+	if eff.vert == "superscript" || eff.vert == "subscript" {
+		vert = eff.vert
+	}
+	emit := func(html string, raw bool) {
+		if kind := cv.fieldKind(); kind != "" && !raw {
+			html = `<span class="doc-field" data-field="` + kind + `">` + html + `</span>`
+		}
+		*pieces = append(*pieces, inlinePiece{css: css, vert: vert, link: link, html: html, raw: raw})
+	}
+	for i := range r.Nodes {
+		c := &r.Nodes[i]
+		switch c.XMLName.Local {
+		case "fldChar":
+			switch c.attr("fldCharType") {
+			case "begin":
+				cv.fields = append(cv.fields, fieldFrame{})
+			case "separate":
+				if n := len(cv.fields); n > 0 {
+					cv.fields[n-1].inResult = true
+				}
+			case "end":
+				if n := len(cv.fields); n > 0 {
+					cv.fields = cv.fields[:n-1]
+				}
+			}
+			continue
+		case "instrText":
+			if n := len(cv.fields); n > 0 && !cv.fields[n-1].inResult {
+				cv.fields[n-1].instr += strings.ToUpper(c.Text)
+				cv.fields[n-1].instr = strings.TrimSpace(cv.fields[n-1].instr)
+			}
+			continue
+		}
+		if cv.fieldHidden() {
+			continue
+		}
+		switch c.XMLName.Local {
+		case "t":
+			t := c.Text
+			if t == "" {
+				continue
+			}
+			if strings.Contains(t, "  ") || strings.Contains(t, "\t") || (strings.HasPrefix(t, " ") && !res.midText) {
+				res.preWrap = true
+			}
+			res.midText = !strings.HasSuffix(t, " ")
+			emit(xmlEscape(t), false)
+		case "tab":
+			emit(`<span class="doc-tab">`+"\t"+`</span>`, false)
+		case "ptab":
+			emit(`<span class="doc-tab">`+"\t"+`</span>`, false)
+		case "br", "cr":
+			if c.attr("type") == "page" {
+				pageBreak()
+				continue
+			}
+			emit("<br>", true)
+			res.midText = false
+		case "noBreakHyphen":
+			emit("\u2011", false)
+		case "softHyphen":
+			emit("\u00ad", false)
+		case "sym":
+			if v, err := strconv.ParseUint(c.attr("char"), 16, 32); err == nil {
+				// symbol fonts map glyphs into the private use area
+				if v >= 0xF000 && v <= 0xF0FF {
+					v -= 0xF000
+				}
+				if v >= 32 {
+					emit(xmlEscape(string(rune(v))), false)
+				}
+			}
+		case "footnoteReference":
+			id := c.attr("id")
+			if onOff2(c.attr("customMarkFollows")) {
+				continue
+			}
+			num, ok := cv.fnNumber[id]
+			if !ok {
+				num = len(cv.fnOrder) + 1
+				cv.fnNumber[id] = num
+				cv.fnOrder = append(cv.fnOrder, id)
+			}
+			// the reference is its own superscript: the run's vertAlign
+			// must not wrap it in a second one
+			*pieces = append(*pieces, inlinePiece{css: css, link: link, raw: true,
+				html: `<sup class="doc-fnref" data-fn="` + xmlEscape(id) + `" contenteditable="false">` + strconv.Itoa(num) + `</sup>`})
+		case "footnoteRef":
+			// the number inside the footnote itself - the editor draws it
+			continue
+		case "drawing":
+			if img := cv.drawing(c, part); img != "" {
+				emit(img, true)
+			}
+		case "pict", "object":
+			if img := cv.vmlImage(c, part); img != "" {
+				emit(img, true)
+			}
+		case "AlternateContent":
+			if ch := c.first("Choice"); ch != nil {
+				if d := ch.first("drawing"); d != nil {
+					if img := cv.drawing(d, part); img != "" {
+						emit(img, true)
+					}
+				}
+			}
+		case "ruby":
+			var texts []string
+			collectText(c.first("rubyBase"), &texts)
+			if s := strings.Join(texts, ""); s != "" {
+				emit(xmlEscape(s), false)
+			}
+		}
+	}
+}
+
+// joinPieces merges adjacent pieces sharing one format into one span
+func joinPieces(pieces []inlinePiece) string {
+	var sb strings.Builder
+	i := 0
+	for i < len(pieces) {
+		// group by link first
+		link := pieces[i].link
+		j := i
+		for j < len(pieces) && pieces[j].link == link {
+			j++
+		}
+		var inner strings.Builder
+		k := i
+		for k < j {
+			css, vert := pieces[k].css, pieces[k].vert
+			m := k
+			var text strings.Builder
+			for m < j && pieces[m].css == css && pieces[m].vert == vert {
+				text.WriteString(pieces[m].html)
+				m++
+			}
+			h := text.String()
+			if css != "" {
+				h = `<span style="` + xmlEscape(css) + `">` + h + `</span>`
+			}
+			switch vert {
+			case "superscript":
+				h = "<sup>" + h + "</sup>"
+			case "subscript":
+				h = "<sub>" + h + "</sub>"
+			}
+			inner.WriteString(h)
+			k = m
+		}
+		if link != "" {
+			sb.WriteString(`<a href="` + xmlEscape(link) + `" style="color:inherit;text-decoration:inherit;">` + inner.String() + `</a>`)
+		} else {
+			sb.WriteString(inner.String())
+		}
+		i = j
+	}
+	return sb.String()
+}
+
+func stripTags(s string) string {
+	return tagRe.ReplaceAllString(s, "")
+}
+
+/* ---------------- pictures ---------------- */
+
+func (cv *docxConv) mediaData(part *docxPartCtx, rid string) (string, bool) {
+	data, ext, ok := cv.mediaBytes(part, rid)
+	if !ok {
+		return "", false
+	}
+	return encodeDataURL(data, ext), true
+}
+
+// mediaBytes resolves an embedded picture to its bytes and image format
+func (cv *docxConv) mediaBytes(part *docxPartCtx, rid string) ([]byte, string, bool) {
+	target, ok := part.rels[rid]
+	if !ok || part.ext[rid] {
+		return nil, "", false
+	}
+	dir := path.Dir(part.name)
+	mediaPath := resolvePartPath(dir, target)
+	data, ok := cv.files[mediaPath]
+	if !ok {
+		return nil, "", false
+	}
+	ext := strings.TrimPrefix(strings.ToLower(path.Ext(mediaPath)), ".")
+	switch ext {
+	case "jpg":
+		ext = "jpeg"
+	case "emf", "wmf", "tif", "tiff":
+		// browsers cannot show these - keep the space, lose the picture
+		return nil, "", false
+	}
+	return data, ext, true
+}
+
+// drawing converts a DrawingML picture (inline or anchored)
+func (cv *docxConv) drawing(n *xnode, part *docxPartCtx) string {
+	holder := n.first("inline")
+	anchor := false
+	if holder == nil {
+		holder = n.first("anchor")
+		anchor = holder != nil
+	}
+	if holder == nil {
+		return ""
+	}
+	var blips []*xnode
+	holder.findAll("blip", &blips)
+	if len(blips) == 0 {
+		return ""
+	}
+	rid := blips[0].attrNS("relationships", "embed")
+	if rid == "" {
+		rid = blips[0].attr("embed")
+	}
+	data, format, ok := cv.mediaBytes(part, rid)
+	if !ok {
+		return ""
+	}
+	wPt, hPt := 0.0, 0.0
+	if ext := holder.first("extent"); ext != nil {
+		wPt = twipsAttr(ext, "cx", 0) / emuPerPt
+		hPt = twipsAttr(ext, "cy", 0) / emuPerPt
+	}
+	// a picture turned by quarter turns or mirrored: bake it into the bitmap
+	turns, flipH, flipV := 0, false, false
+	if spPr := findFirst(holder, "spPr"); spPr != nil {
+		if xf := spPr.first("xfrm"); xf != nil {
+			if t, ok := pictureTurns(twipsAttr(xf, "rot", 0)); ok {
+				turns = t
+			}
+			flipH, flipV = onOff2(xf.attr("flipH")), onOff2(xf.attr("flipV"))
+		}
+	}
+	if turns != 0 || flipH || flipV {
+		if d, f, ok := orientPicture(data, format, turns, flipH, flipV); ok {
+			data, format = d, f
+			if turns%2 == 1 {
+				wPt, hPt = hPt, wPt
+			}
+		} else {
+			turns, flipH, flipV = 0, false, false
+		}
+	}
+	src := encodeDataURL(data, format)
+	var css []string
+	if wPt > 0 && hPt > 0 {
+		css = append(css, "width:"+ptStr(wPt), "height:"+ptStr(hPt))
+	}
+	// crop: srcRect is in thousandths of a percent of the source
+	var rects []*xnode
+	holder.findAll("srcRect", &rects)
+	if len(rects) > 0 {
+		t := twipsAttr(rects[0], "t", 0) / 1000
+		r := twipsAttr(rects[0], "r", 0) / 1000
+		b := twipsAttr(rects[0], "b", 0) / 1000
+		l := twipsAttr(rects[0], "l", 0) / 1000
+		if t != 0 || r != 0 || b != 0 || l != 0 {
+			o := orientInsets([4]float64{t, r, b, l}, turns, flipH, flipV)
+			t, r, b, l = o[0], o[1], o[2], o[3]
+			css = append(css, fmt.Sprintf("object-fit:fill;object-view-box:inset(%s%% %s%% %s%% %s%%)",
+				trimFloat(round3(t)), trimFloat(round3(r)), trimFloat(round3(b)), trimFloat(round3(l))))
+		}
+	}
+	// a picture outline
+	borderPt := 0.0
+	if spPr := findFirst(holder, "spPr"); spPr != nil {
+		if ln := spPr.first("ln"); ln != nil && ln.first("noFill") == nil {
+			if fill := ln.first("solidFill"); fill != nil {
+				col := "000000"
+				if c := fill.first("srgbClr"); c != nil && len(c.attr("val")) == 6 {
+					col = strings.ToLower(c.attr("val"))
+				}
+				w := twipsAttr(ln, "w", 9525) / emuPerPt
+				css = append(css, "border:"+ptStr(w)+" solid #"+col)
+				borderPt = w
+			}
+		}
+	}
+	if !anchor {
+		// room beside the frame: the effect extent beyond the outline (what
+		// the docx writer stores a picture's side margins in), else Google
+		// Docs' own 1.5pt (the space above and below a picture is the
+		// layout engine's business - see pictureLines in docs_layout.js)
+		ml, mr := 0.0, 0.0
+		if ee := holder.first("effectExtent"); ee != nil && turns%2 == 0 {
+			// (a quarter-turned frame keeps its turn in the extent instead)
+			ml = twipsAttr(ee, "l", 0)/emuPerPt - borderPt
+			mr = twipsAttr(ee, "r", 0)/emuPerPt - borderPt
+		}
+		if ml < 0.05 && mr < 0.05 && cv.gdocs {
+			ml, mr = 1.5, 1.5
+		}
+		if ml >= 0.05 {
+			css = append(css, "margin-left:"+ptStr(ml))
+		}
+		if mr >= 0.05 {
+			css = append(css, "margin-right:"+ptStr(mr))
+		}
+	}
+	alt := ""
+	if dp := holder.first("docPr"); dp != nil {
+		alt = dp.attr("descr")
+	}
+	cls := ""
+	if anchor {
+		x, y := 0.0, 0.0
+		if ph := holder.first("positionH"); ph != nil {
+			if off := ph.first("posOffset"); off != nil {
+				if v, err := strconv.ParseFloat(strings.TrimSpace(off.Text), 64); err == nil {
+					x = v / emuPerPt
+				}
+			}
+			switch ph.attr("relativeFrom") {
+			case "page":
+				x -= cv.marginL
+			}
+			if al := ph.first("align"); al != nil {
+				switch strings.TrimSpace(al.Text) {
+				case "center":
+					x = (cv.textW - wPt) / 2
+				case "right":
+					x = cv.textW - wPt
+				}
+			}
+		}
+		if pv := holder.first("positionV"); pv != nil {
+			if off := pv.first("posOffset"); off != nil && (pv.attr("relativeFrom") == "paragraph" || pv.attr("relativeFrom") == "line") {
+				if v, err := strconv.ParseFloat(strings.TrimSpace(off.Text), 64); err == nil {
+					y = v / emuPerPt
+				}
+			}
+		}
+		square := holder.first("wrapSquare") != nil || holder.first("wrapTight") != nil ||
+			holder.first("wrapThrough") != nil
+		if square {
+			// text flows around it: float to the side it sits on
+			side := "left"
+			if x+wPt/2 > cv.textW/2 {
+				side = "right"
+			}
+			css = append(css, "float:"+side, "max-width:none")
+			if side == "left" && x > 0 {
+				css = append(css, "margin-left:"+ptStr(x))
+			}
+			css = append(css, "margin-right:9pt", "margin-left:9pt")
+		} else {
+			css = append(css, "display:block", "max-width:none")
+			if x != 0 {
+				css = append(css, "margin-left:"+ptStr(x))
+			}
+			if y != 0 {
+				css = append(css, "margin-top:"+ptStr(y))
+			}
+		}
+		cls = ` class="doc-anchor"`
+	}
+	out := `<img src="` + src + `"` + cls
+	if alt != "" {
+		out += ` alt="` + xmlEscape(alt) + `"`
+	}
+	if len(css) > 0 {
+		out += ` style="` + strings.Join(css, ";") + `;"`
+	}
+	return out + ">"
+}
+
+func findFirst(n *xnode, local string) *xnode {
+	var all []*xnode
+	n.findAll(local, &all)
+	if len(all) == 0 {
+		return nil
+	}
+	return all[0]
+}
+
+// vmlImage converts a legacy VML picture (w:pict / w:object)
+func (cv *docxConv) vmlImage(n *xnode, part *docxPartCtx) string {
+	var datas []*xnode
+	n.findAll("imagedata", &datas)
+	if len(datas) == 0 {
+		return ""
+	}
+	rid := datas[0].attrNS("relationships", "id")
+	if rid == "" {
+		rid = datas[0].attr("id")
+	}
+	src, ok := cv.mediaData(part, rid)
+	if !ok {
+		return ""
+	}
+	style := ""
+	var shapes []*xnode
+	n.findAll("shape", &shapes)
+	if len(shapes) > 0 {
+		st := shapes[0].attr("style")
+		w := cssLengthPt(styleProp(st, "width"))
+		h := cssLengthPt(styleProp(st, "height"))
+		if w > 0 && h > 0 {
+			style = ` style="width:` + ptStr(w) + `;height:` + ptStr(h) + `;"`
+		}
+	}
+	return `<img src="` + src + `"` + style + `>`
+}
+
+// cssLengthPt converts a CSS length (pt, px, in, cm, mm) to points
+func cssLengthPt(s string) float64 {
+	s = strings.TrimSpace(strings.ToLower(s))
+	units := []struct {
+		suf string
+		k   float64
+	}{{"pt", 1}, {"px", 0.75}, {"in", 72}, {"cm", 72 / 2.54}, {"mm", 72 / 25.4}}
+	for _, u := range units {
+		if strings.HasSuffix(s, u.suf) {
+			if v, err := strconv.ParseFloat(strings.TrimSuffix(s, u.suf), 64); err == nil {
+				return v * u.k
+			}
+		}
+	}
+	if v, err := strconv.ParseFloat(s, 64); err == nil {
+		return v * 0.75
+	}
+	return 0
+}
+
+/* ---------------- tables ---------------- */
+
+type docxCell struct {
+	node    *xnode
+	tcPr    *xnode
+	gridCol int
+	span    int
+	vMerge  string // restart | continue | ""
+	rowspan int
+}
+
+func (cv *docxConv) table(tbl *xnode, part *docxPartCtx, spanAll bool) string {
+	direct := parseTblPr(tbl.first("tblPr"))
+	tp := *cv.ss.tableStyle(direct.style)
+	tp.merge(direct)
+
+	// grid
+	var grid []float64
+	if g := tbl.first("tblGrid"); g != nil {
+		for _, gc := range g.all("gridCol") {
+			grid = append(grid, twipsAttr(gc, "w", 0)/20)
+		}
+	}
+
+	// rows and cells with their grid positions
+	type rowInfo struct {
+		node  *xnode
+		cells []*docxCell
+	}
+	var rows []*rowInfo
+	for i := range tbl.Nodes {
+		tr := &tbl.Nodes[i]
+		if tr.XMLName.Local != "tr" {
+			continue
+		}
+		ri := &rowInfo{node: tr}
+		col := 0
+		if trPr := tr.first("trPr"); trPr != nil {
+			if gb := trPr.first("gridBefore"); gb != nil {
+				if v, err := strconv.Atoi(gb.attr("val")); err == nil {
+					col += v
+				}
+			}
+		}
+		var addCells func(parent *xnode)
+		addCells = func(parent *xnode) {
+			for j := range parent.Nodes {
+				tc := &parent.Nodes[j]
+				switch tc.XMLName.Local {
+				case "tc":
+					c := &docxCell{node: tc, tcPr: tc.first("tcPr"), gridCol: col, span: 1, rowspan: 1}
+					if c.tcPr != nil {
+						if gs := c.tcPr.first("gridSpan"); gs != nil {
+							if v, err := strconv.Atoi(gs.attr("val")); err == nil && v > 1 {
+								c.span = v
+							}
+						}
+						if vm := c.tcPr.first("vMerge"); vm != nil {
+							c.vMerge = vm.attr("val")
+							if c.vMerge == "" {
+								c.vMerge = "continue"
+							}
+						}
+					}
+					ri.cells = append(ri.cells, c)
+					col += c.span
+				case "sdt":
+					if sc := tc.first("sdtContent"); sc != nil {
+						addCells(sc)
+					}
+				case "customXml":
+					addCells(tc)
+				}
+			}
+		}
+		addCells(tr)
+		rows = append(rows, ri)
+	}
+	// vertical merges -> rowspan on the restarting cell
+	for r, ri := range rows {
+		for _, c := range ri.cells {
+			if c.vMerge != "restart" {
+				continue
+			}
+			for r2 := r + 1; r2 < len(rows); r2++ {
+				found := false
+				for _, c2 := range rows[r2].cells {
+					if c2.gridCol == c.gridCol && c2.vMerge == "continue" {
+						found = true
+						c.rowspan++
+						break
+					}
+				}
+				if !found {
+					break
+				}
+			}
+		}
+	}
+	nCols := len(grid)
+	for _, ri := range rows {
+		n := 0
+		for _, c := range ri.cells {
+			n = maxInt(n, c.gridCol+c.span)
+		}
+		nCols = maxInt(nCols, n)
+	}
+	for len(grid) < nCols {
+		grid = append(grid, cv.textW/float64(maxInt(nCols, 1)))
+	}
+	totalW := 0.0
+	for _, w := range grid {
+		totalW += w
+	}
+	if tp.width.set && tp.widthType == "dxa" && tp.width.v > 0 && totalW == 0 {
+		totalW = tp.width.v / 20
+	}
+
+	var css []string
+	if totalW > 0 {
+		// the grid decides the columns, not the content (Word's fixed
+		// layout, and what Google Docs always does)
+		css = append(css, "width:"+ptStr(totalW), "table-layout:fixed")
+	}
+	switch tp.jc {
+	case "center":
+		css = append(css, "margin-left:auto", "margin-right:auto")
+	case "right", "end":
+		css = append(css, "margin-left:auto", "margin-right:0")
+	default:
+		if tp.ind.set && tp.ind.v != 0 {
+			css = append(css, "margin-left:"+ptStr(tp.ind.v/20))
+		}
+	}
+	cls := "of-table"
+	if spanAll {
+		cls += " col-span-all"
+	}
+	// data-docx: the table lays out as Word's (no spacing around it) -
+	// except one the editor made, which keeps the editor's own
+	docxAttr := ` data-docx="1"`
+	if pr := tbl.first("tblPr"); pr != nil {
+		if st := pr.first("tblStyle"); st != nil && st.attr("val") == editorTableStyle {
+			docxAttr = ""
+		}
+	}
+	var sb strings.Builder
+	sb.WriteString(`<table class="` + cls + `"` + docxAttr + ` style="` + strings.Join(css, ";") + `;">`)
+	if len(grid) > 0 {
+		sb.WriteString("<colgroup>")
+		for _, w := range grid {
+			sb.WriteString(`<col style="width:` + ptStr(w) + `">`)
+		}
+		sb.WriteString("</colgroup>")
+	}
+	sb.WriteString("<tbody>")
+	for r, ri := range rows {
+		trStyle := ""
+		trData := ""
+		if trPr := ri.node.first("trPr"); trPr != nil {
+			if th := trPr.first("trHeight"); th != nil {
+				if v := numAttr(th, "val"); v.set && v.v > 0 {
+					trStyle = ` style="height:` + ptStr(v.v/20) + `;"`
+					if th.attr("hRule") == "exact" {
+						trData += ` data-exact="1"`
+					}
+				}
+			}
+			if onOff(trPr.first("cantSplit")).v {
+				trData += ` data-cant-split="1"`
+			}
+			if onOff(trPr.first("tblHeader")).v {
+				trData += ` data-header-row="1"`
 			}
 			}
-			sb.WriteString("<td" + tdStyle + ">" + inner + "</td>")
+		}
+		sb.WriteString("<tr" + trStyle + trData + ">")
+		for _, c := range ri.cells {
+			if c.vMerge == "continue" {
+				continue
+			}
+			sb.WriteString(cv.cell(c, tp, r, len(rows), nCols, c.rowspan, part))
 		}
 		}
 		sb.WriteString("</tr>")
 		sb.WriteString("</tr>")
 	}
 	}
-	sb.WriteString("</table>")
+	sb.WriteString("</tbody></table>")
 	return sb.String()
 	return sb.String()
 }
 }
+
+func (cv *docxConv) cell(c *docxCell, tp docxTblPr, row, nRows, nCols, rowspan int, part *docxPartCtx) string {
+	var tcBorders map[string]docxBorder
+	var tcMar map[string]optNum
+	shd, vAlign := "", ""
+	if c.tcPr != nil {
+		tcBorders = parseBorderSet(c.tcPr.first("tcBorders"))
+		tcMar = parseMarginSet(c.tcPr.first("tcMar"))
+		if s := c.tcPr.first("shd"); s != nil {
+			if f := strings.ToLower(s.attr("fill")); len(f) == 6 {
+				shd = f
+			}
+		}
+		if va := c.tcPr.first("vAlign"); va != nil {
+			vAlign = va.attr("val")
+		}
+	}
+	lastRow := row+rowspan >= nRows
+	firstCol := c.gridCol == 0
+	lastCol := c.gridCol+c.span >= nCols
+	side := func(name, outer, inner string, isOuter bool) string {
+		if b, ok := tcBorders[name]; ok {
+			if !b.visible() {
+				return "none"
+			}
+			return borderCSS(b)
+		}
+		key := inner
+		if isOuter {
+			key = outer
+		}
+		if b, ok := tp.borders[key]; ok && b.visible() {
+			return borderCSS(b)
+		}
+		return "none"
+	}
+	var css []string
+	css = append(css,
+		"border-top:"+side("top", "top", "insideH", row == 0),
+		"border-right:"+side("right", "right", "insideV", lastCol),
+		"border-bottom:"+side("bottom", "bottom", "insideH", lastRow),
+		"border-left:"+side("left", "left", "insideV", firstCol))
+	mar := func(name string, def float64) float64 {
+		if v, ok := tcMar[name]; ok {
+			return v.v / 20
+		}
+		if v, ok := tp.cellMar[name]; ok {
+			return v.v / 20
+		}
+		return def
+	}
+	// Word's defaults when nothing states a margin: 0.08" left and right
+	css = append(css, "padding:"+ptStr(mar("top", 0))+" "+ptStr(mar("right", 5.4))+" "+
+		ptStr(mar("bottom", 0))+" "+ptStr(mar("left", 5.4)))
+	switch vAlign {
+	case "center":
+		css = append(css, "vertical-align:middle")
+	case "bottom":
+		css = append(css, "vertical-align:bottom")
+	default:
+		css = append(css, "vertical-align:top")
+	}
+	if shd != "" && shd != "auto" {
+		css = append(css, "background-color:#"+shd)
+	}
+	attrs := ""
+	if c.span > 1 {
+		attrs += ` colspan="` + strconv.Itoa(c.span) + `"`
+	}
+	if rowspan > 1 {
+		attrs += ` rowspan="` + strconv.Itoa(rowspan) + `"`
+	}
+	inner := cv.blocks(c.node, part, blockCtx{cell: true})
+	if inner == "" {
+		inner = "<p><br></p>"
+	}
+	return "<td" + attrs + ` style="` + strings.Join(css, ";") + `;">` + inner + "</td>"
+}

+ 435 - 0
src/mod/office/docx_rich_test.go

@@ -0,0 +1,435 @@
+package office
+
+import (
+	"archive/zip"
+	"bytes"
+	"image"
+	"image/color"
+	"image/png"
+	"strings"
+	"testing"
+)
+
+const wNS = `xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" ` +
+	`xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships" ` +
+	`xmlns:wp="http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing" ` +
+	`xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main" ` +
+	`xmlns:pic="http://schemas.openxmlformats.org/drawingml/2006/picture"`
+
+// testDocx zips a minimal package: the body XML plus any extra parts
+func testDocx(t *testing.T, body string, parts map[string]string) []byte {
+	t.Helper()
+	var buf bytes.Buffer
+	zw := zip.NewWriter(&buf)
+	all := map[string]string{
+		"word/document.xml": `<?xml version="1.0" encoding="UTF-8"?><w:document ` + wNS + `><w:body>` + body +
+			`<w:sectPr><w:pgSz w:w="11906" w:h="16838"/><w:pgMar w:top="1440" w:right="1440" w:bottom="1440" w:left="1440" w:header="720" w:footer="720"/></w:sectPr></w:body></w:document>`,
+	}
+	for k, v := range parts {
+		all[k] = v
+	}
+	for name, content := range all {
+		f, err := zw.Create(name)
+		if err != nil {
+			t.Fatalf("zip create: %v", err)
+		}
+		if _, err := f.Write([]byte(content)); err != nil {
+			t.Fatalf("zip write: %v", err)
+		}
+	}
+	if err := zw.Close(); err != nil {
+		t.Fatalf("zip close: %v", err)
+	}
+	return buf.Bytes()
+}
+
+func parseTestDocx(t *testing.T, body string, parts map[string]string) *Document {
+	t.Helper()
+	doc, err := ParseDocx(testDocx(t, body, parts))
+	if err != nil {
+		t.Fatalf("ParseDocx: %v", err)
+	}
+	return doc
+}
+
+func TestOnOff(t *testing.T) {
+	tests := []struct {
+		xml     string
+		set, on bool
+	}{
+		{`<w:b/>`, true, true},
+		{`<w:b w:val="1"/>`, true, true},
+		{`<w:b w:val="true"/>`, true, true},
+		{`<w:b w:val="0"/>`, true, false},
+		{`<w:b w:val="false"/>`, true, false},
+		{`<w:b w:val="off"/>`, true, false},
+	}
+	for _, tc := range tests {
+		n, err := parseXMLTree([]byte(`<w:rPr ` + wNS + `>` + tc.xml + `</w:rPr>`))
+		if err != nil {
+			t.Fatalf("%s: %v", tc.xml, err)
+		}
+		got := onOff(n.first("b"))
+		if got.set != tc.set || got.v != tc.on {
+			t.Errorf("onOff(%s) = %+v, want set=%v on=%v", tc.xml, got, tc.set, tc.on)
+		}
+	}
+	if got := onOff(nil); got.set {
+		t.Errorf("onOff(nil) = %+v, want unset", got)
+	}
+}
+
+func TestFormatListNumber(t *testing.T) {
+	tests := []struct {
+		v    int
+		fmt  string
+		want string
+	}{
+		{3, "decimal", "3"},
+		{1, "lowerLetter", "a"},
+		{28, "lowerLetter", "bb"},
+		{2, "upperLetter", "B"},
+		{4, "lowerRoman", "iv"},
+		{1994, "upperRoman", "MCMXCIV"},
+		{7, "decimalZero", "07"},
+		{12, "decimalZero", "12"},
+		{5, "bullet", ""},
+	}
+	for _, tc := range tests {
+		if got := formatListNumber(tc.v, tc.fmt); got != tc.want {
+			t.Errorf("formatListNumber(%d, %s) = %q, want %q", tc.v, tc.fmt, got, tc.want)
+		}
+	}
+}
+
+func TestPictureTurns(t *testing.T) {
+	tests := []struct {
+		rot   float64
+		turns int
+		ok    bool
+	}{
+		{0, 0, true},
+		{5400000, 1, true},
+		{10800000, 2, true},
+		{16200000, 3, true},
+		{-5400000, 3, true},
+		{21600000, 0, true},
+		{5430000, 1, true}, // within a degree
+		{2700000, 0, false},
+	}
+	for _, tc := range tests {
+		turns, ok := pictureTurns(tc.rot)
+		if ok != tc.ok || (ok && turns != tc.turns) {
+			t.Errorf("pictureTurns(%v) = %d,%v want %d,%v", tc.rot, turns, ok, tc.turns, tc.ok)
+		}
+	}
+}
+
+func TestOrientInsets(t *testing.T) {
+	in := [4]float64{1, 2, 3, 4} // top right bottom left
+	tests := []struct {
+		name         string
+		turns        int
+		flipH, flipV bool
+		want         [4]float64
+	}{
+		{"none", 0, false, false, [4]float64{1, 2, 3, 4}},
+		{"quarter", 1, false, false, [4]float64{4, 1, 2, 3}},
+		{"half", 2, false, false, [4]float64{3, 4, 1, 2}},
+		{"three quarters", 3, false, false, [4]float64{2, 3, 4, 1}},
+		{"mirror", 0, true, false, [4]float64{1, 4, 3, 2}},
+		{"flip", 0, false, true, [4]float64{3, 2, 1, 4}},
+	}
+	for _, tc := range tests {
+		if got := orientInsets(in, tc.turns, tc.flipH, tc.flipV); got != tc.want {
+			t.Errorf("%s: orientInsets = %v, want %v", tc.name, got, tc.want)
+		}
+	}
+}
+
+func TestOrientPixels(t *testing.T) {
+	// a 3x2 picture with a red top-left pixel
+	src := image.NewNRGBA(image.Rect(0, 0, 3, 2))
+	red := color.NRGBA{R: 255, A: 255}
+	src.SetNRGBA(0, 0, red)
+	tests := []struct {
+		name         string
+		turns        int
+		flipH, flipV bool
+		w, h, rx, ry int
+	}{
+		{"quarter turn", 1, false, false, 2, 3, 1, 0},
+		{"half turn", 2, false, false, 3, 2, 2, 1},
+		{"three quarters", 3, false, false, 2, 3, 0, 2},
+		{"mirror", 0, true, false, 3, 2, 2, 0},
+		{"flip", 0, false, true, 3, 2, 0, 1},
+	}
+	for _, tc := range tests {
+		out := orientPixels(src, tc.turns, tc.flipH, tc.flipV)
+		if out.Bounds().Dx() != tc.w || out.Bounds().Dy() != tc.h {
+			t.Errorf("%s: size %v, want %dx%d", tc.name, out.Bounds().Size(), tc.w, tc.h)
+			continue
+		}
+		if out.NRGBAAt(tc.rx, tc.ry) != red {
+			t.Errorf("%s: red pixel not at (%d,%d)", tc.name, tc.rx, tc.ry)
+		}
+	}
+}
+
+func TestDocxReaderRunAndParagraphRules(t *testing.T) {
+	tests := []struct {
+		name, body string
+		want       []string
+		dont       []string
+	}{
+		{
+			name: "explicit off toggle",
+			body: `<w:p><w:r><w:rPr><w:b w:val="0"/></w:rPr><w:t>plain</w:t></w:r></w:p>`,
+			want: []string{`>plain</p>`},
+			dont: []string{"font-weight:700"},
+		},
+		{
+			name: "spacing before and after",
+			body: `<w:p><w:pPr><w:spacing w:before="240" w:after="120"/></w:pPr><w:r><w:t>x</w:t></w:r></w:p>`,
+			want: []string{"padding-top:12pt;margin-bottom:6pt;"},
+		},
+		{
+			name: "exact line height",
+			body: `<w:p><w:pPr><w:spacing w:line="360" w:lineRule="exact"/></w:pPr><w:r><w:t>x</w:t></w:r></w:p>`,
+			want: []string{`data-lsexact="18pt"`},
+		},
+		{
+			name: "a leading space inside the text collapses as usual",
+			body: `<w:p><w:r><w:t xml:space="preserve">one</w:t></w:r><w:r><w:rPr><w:b/></w:rPr><w:t xml:space="preserve"> two</w:t></w:r></w:p>`,
+			dont: []string{"pre-wrap"},
+		},
+		{
+			name: "a leading space at the start is kept",
+			body: `<w:p><w:r><w:t xml:space="preserve"> lead</w:t></w:r></w:p>`,
+			want: []string{"white-space:pre-wrap"},
+		},
+		{
+			name: "right tab stop with leader",
+			body: `<w:p><w:pPr><w:tabs><w:tab w:val="right" w:leader="dot" w:pos="9026"/></w:tabs></w:pPr><w:r><w:t>a</w:t></w:r><w:r><w:tab/><w:t>1</w:t></w:r></w:p>`,
+			want: []string{`data-tabs="right:451.3:dot"`, `<span class="doc-tab">` + "\t" + `</span>1`},
+		},
+		{
+			name: "page break splits the paragraph",
+			body: `<w:p><w:r><w:t>before</w:t></w:r><w:r><w:br w:type="page"/></w:r><w:r><w:t>after</w:t></w:r></w:p>`,
+			want: []string{`>before</p><div class="doc-pagebreak"`, `>after</p>`},
+		},
+		{
+			name: "non-breaking and soft hyphens",
+			body: `<w:p><w:r><w:t>a</w:t><w:noBreakHyphen/><w:t>b</w:t><w:softHyphen/><w:t>c</w:t></w:r></w:p>`,
+			want: []string{"a\u2011b\u00adc"},
+		},
+	}
+	for _, tc := range tests {
+		doc := parseTestDocx(t, tc.body, nil)
+		for _, w := range tc.want {
+			if !strings.Contains(doc.HTML, w) {
+				t.Errorf("%s: want %q in %s", tc.name, w, doc.HTML)
+			}
+		}
+		for _, d := range tc.dont {
+			if strings.Contains(doc.HTML, d) {
+				t.Errorf("%s: did not want %q in %s", tc.name, d, doc.HTML)
+			}
+		}
+	}
+}
+
+func TestDocxReaderFootnotes(t *testing.T) {
+	body := `<w:p><w:r><w:t>text</w:t></w:r><w:r><w:rPr><w:vertAlign w:val="superscript"/></w:rPr><w:footnoteReference w:id="7"/></w:r></w:p>`
+	notes := `<?xml version="1.0" encoding="UTF-8"?><w:footnotes ` + wNS + `>` +
+		`<w:footnote w:type="separator" w:id="-1"><w:p><w:r><w:separator/></w:r></w:p></w:footnote>` +
+		`<w:footnote w:id="7"><w:p><w:r><w:footnoteRef/></w:r><w:r><w:t xml:space="preserve"> the note</w:t></w:r></w:p></w:footnote></w:footnotes>`
+	doc := parseTestDocx(t, body, map[string]string{"word/footnotes.xml": notes})
+	if !strings.Contains(doc.HTML, `<sup class="doc-fnref" data-fn="7" contenteditable="false">1</sup>`) {
+		t.Errorf("footnote reference not imported: %s", doc.HTML)
+	}
+	if strings.Count(doc.HTML, "<sup") != 1 {
+		t.Errorf("footnote reference nested in another superscript: %s", doc.HTML)
+	}
+	if len(doc.Footnotes) != 1 || doc.Footnotes[0].ID != "7" || !strings.Contains(doc.Footnotes[0].HTML, "the note") {
+		t.Errorf("footnotes = %+v", doc.Footnotes)
+	}
+}
+
+func TestDocxReaderRotatedPicture(t *testing.T) {
+	img := image.NewNRGBA(image.Rect(0, 0, 4, 2))
+	var pngBuf bytes.Buffer
+	if err := png.Encode(&pngBuf, img); err != nil {
+		t.Fatalf("png: %v", err)
+	}
+	rels := `<?xml version="1.0" encoding="UTF-8"?><Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">` +
+		`<Relationship Id="rIdImg" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="media/p.png"/></Relationships>`
+	picture := func(rot string) string {
+		return `<w:p><w:r><w:drawing><wp:inline><wp:extent cx="508000" cy="254000"/>` +
+			`<a:graphic><a:graphicData><pic:pic><pic:blipFill><a:blip r:embed="rIdImg"/>` +
+			`<a:srcRect t="10000"/></pic:blipFill><pic:spPr><a:xfrm ` + rot + `><a:ext cx="508000" cy="254000"/></a:xfrm></pic:spPr>` +
+			`</pic:pic></a:graphicData></a:graphic></wp:inline></w:drawing></w:r></w:p>`
+	}
+	tests := []struct {
+		name, rot, want string
+	}{
+		{"upright", ``, "width:40pt;height:20pt;object-fit:fill;object-view-box:inset(10% 0% 0% 0%)"},
+		{"quarter turn", `rot="5400000"`, "width:20pt;height:40pt;object-fit:fill;object-view-box:inset(0% 10% 0% 0%)"},
+	}
+	for _, tc := range tests {
+		zipData := testDocx(t, picture(tc.rot), map[string]string{
+			"word/_rels/document.xml.rels": rels,
+		})
+		// the media part is binary: add it by rebuilding with the bytes
+		doc := parseDocxWithMedia(t, zipData, "word/media/p.png", pngBuf.Bytes())
+		if !strings.Contains(doc.HTML, tc.want) {
+			t.Errorf("%s: want %q in %.400s", tc.name, tc.want, doc.HTML)
+		}
+	}
+}
+
+// parseDocxWithMedia adds one binary part to a test package and parses it
+func parseDocxWithMedia(t *testing.T, zipData []byte, name string, data []byte) *Document {
+	t.Helper()
+	zr, err := zip.NewReader(bytes.NewReader(zipData), int64(len(zipData)))
+	if err != nil {
+		t.Fatalf("zip: %v", err)
+	}
+	var buf bytes.Buffer
+	zw := zip.NewWriter(&buf)
+	for _, f := range zr.File {
+		rc, err := f.Open()
+		if err != nil {
+			t.Fatalf("zip open: %v", err)
+		}
+		w, _ := zw.Create(f.Name)
+		var b bytes.Buffer
+		if _, err := b.ReadFrom(rc); err != nil {
+			t.Fatalf("zip read: %v", err)
+		}
+		rc.Close()
+		w.Write(b.Bytes())
+	}
+	w, _ := zw.Create(name)
+	w.Write(data)
+	if err := zw.Close(); err != nil {
+		t.Fatalf("zip close: %v", err)
+	}
+	doc, err := ParseDocx(buf.Bytes())
+	if err != nil {
+		t.Fatalf("ParseDocx: %v", err)
+	}
+	return doc
+}
+
+func TestDocxRichRoundTrip(t *testing.T) {
+	// the editor's own model survives docx -> model unchanged
+	tests := []struct {
+		name, html string
+	}{
+		{"spacing and indents", `<p style="padding-top:6pt;margin-bottom:10pt;margin-left:36pt;text-indent:-18pt;">x</p>`},
+		{"exact line height", `<p data-lsexact="14pt">x</p>`},
+		{"line spacing", `<p data-ls="1.5">x</p>`},
+		{"keep with next", `<p data-keep-next="1">x</p>`},
+		{"tab stops", `<p data-tabs="right:451.3:dot">a<span class="doc-tab">` + "\t" + `</span>1</p>`},
+		{"footnote reference", `<p>a<sup class="doc-fnref" data-fn="1" contenteditable="false">1</sup></p>`},
+		{"page field", `<p><span class="doc-field" data-field="PAGE">1</span></p>`},
+		{"shaded paragraph", `<p style="background-color:#ffe599;">x</p>`},
+		{"trailing line break", `<p>x<br><br></p>`},
+		{"horizontal rule", `<p>a</p><hr><p>b</p>`},
+		{"bordered paragraph", `<p style="margin-top:6pt;margin-left:10pt;border-left:1.5pt solid #ff0000;padding-left:4pt;">x</p>`},
+	}
+	for _, tc := range tests {
+		src := &Document{HTML: tc.html, Footnotes: []Footnote{{ID: "1", HTML: "<p>note</p>"}}}
+		data, err := BuildDocx(src)
+		if err != nil {
+			t.Fatalf("%s: BuildDocx: %v", tc.name, err)
+		}
+		back, err := ParseDocx(data)
+		if err != nil {
+			t.Fatalf("%s: ParseDocx: %v", tc.name, err)
+		}
+		if !strings.Contains(back.HTML, tc.html) {
+			t.Errorf("%s: want %s, got %s", tc.name, tc.html, back.HTML)
+		}
+	}
+}
+
+func TestDocxWriterEditorBoxes(t *testing.T) {
+	tests := []struct {
+		name, html string
+		want       []string
+	}{
+		{
+			// docs.css: 3px rule, 12px padding, 8pt + 2pt either side
+			name: "blockquote",
+			html: `<blockquote>quoted</blockquote>`,
+			want: []string{`<w:left w:val="single" w:sz="18" w:space="9" w:color="C3C7CC"/>`,
+				`<w:spacing w:before="200" w:after="200"`, `<w:ind w:left="225"`, `<w:color w:val="5F6368"/>`},
+		},
+		{
+			// a 1px frame with 10px 12px inside: the half point w:space
+			// cannot hold goes to the spacing outside
+			name: "code block",
+			html: `<pre>code</pre>`,
+			want: []string{`<w:top w:val="single" w:sz="6" w:space="8" w:color="E2E5E9"/>`,
+				`<w:left w:val="single" w:sz="6" w:space="9" w:color="E2E5E9"/>`,
+				`<w:spacing w:before="150" w:after="150"`, `<w:ind w:left="195" w:right="195"`,
+				`<w:shd w:val="clear" w:color="auto" w:fill="F1F3F4"/>`, `w:ascii="Consolas"`},
+		},
+		{
+			name: "an indent blockquote from the browser keeps no rule",
+			html: `<blockquote style="margin: 0 0 0 40px; border: none; padding: 0px;"><p>x</p></blockquote>`,
+			want: []string{`<w:ind w:left="600"`},
+		},
+		{
+			name: "header cell is centred",
+			html: `<table class="of-table"><tbody><tr><th>H</th></tr></tbody></table>`,
+			want: []string{`<w:tblStyle w:val="ArozEditorTable"/>`, `<w:jc w:val="center"/>`},
+		},
+		{
+			name: "imported table stays a Word table",
+			html: `<table class="of-table" data-docx="1"><tbody><tr><td>c</td></tr></tbody></table>`,
+			want: []string{`<w:tblStyle w:val="TableGrid"/>`},
+		},
+		{
+			name: "rule",
+			html: `<hr>`,
+			want: []string{`<w:pStyle w:val="ArozHorizontalRule"/>`, `w:line="20" w:lineRule="exact"`},
+		},
+	}
+	for _, tc := range tests {
+		data, err := BuildDocx(&Document{HTML: tc.html})
+		if err != nil {
+			t.Fatalf("%s: BuildDocx: %v", tc.name, err)
+		}
+		body := string(zipPart(t, data, "word/document.xml"))
+		for _, w := range tc.want {
+			if !strings.Contains(body, w) {
+				t.Errorf("%s: want %s in %s", tc.name, w, body)
+			}
+		}
+	}
+	// the indent blockquote draws no rule
+	data, _ := BuildDocx(&Document{HTML: tests[2].html})
+	if body := string(zipPart(t, data, "word/document.xml")); strings.Contains(body, "<w:pBdr>") {
+		t.Errorf("border:none blockquote got a rule: %s", body)
+	}
+}
+
+func TestDocxPlainHeaderStaysPlain(t *testing.T) {
+	data, err := BuildDocx(&Document{HTML: "<p>x</p>", Header: "Top & tail", Footer: "Bottom"})
+	if err != nil {
+		t.Fatalf("BuildDocx: %v", err)
+	}
+	back, err := ParseDocx(data)
+	if err != nil {
+		t.Fatalf("ParseDocx: %v", err)
+	}
+	if back.Header != "Top & tail" || back.HeaderHTML != "" {
+		t.Errorf("header = %q / %q, want plain text", back.Header, back.HeaderHTML)
+	}
+	if back.Footer != "Bottom" || back.FooterHTML != "" || back.PageNumbers {
+		t.Errorf("footer = %q / %q pn=%v, want plain text", back.Footer, back.FooterHTML, back.PageNumbers)
+	}
+}

+ 26 - 14
src/mod/office/docx_test.go

@@ -87,24 +87,28 @@ func TestDocxRoundtrip(t *testing.T) {
 	}{
 	}{
 		{"title", "doc-title"},
 		{"title", "doc-title"},
 		{"title text", "My Report"},
 		{"title text", "My Report"},
-		{"heading", "<h2>Section &amp; Chapter</h2>"},
-		{"bold", "<b>bold</b>"},
-		{"italic", "<i>italic</i>"},
-		{"color+size", "color:#cc0000"},
-		{"font size", "font-size:20px"},
+		{"heading", ">Section &amp; Chapter</h2>"},
+		{"heading size", "font-size:16pt;font-weight:700"},
+		{"bold", `<span style="font-weight:700;">bold</span>`},
+		{"italic", `<span style="font-style:italic;">italic</span>`},
+		{"color+size", "font-size:15pt;color:#cc0000;"},
 		{"center align", "text-align:center"},
 		{"center align", "text-align:center"},
 		{"link", `href="https://example.com/x?a=1"`},
 		{"link", `href="https://example.com/x?a=1"`},
-		{"line break", "<br>"},
-		{"bullet list", "<ul><li>alpha</li><li>beta</li></ul>"},
-		{"numbered list", "<ol><li>one</li><li>two</li></ol>"},
-		{"blockquote", "<blockquote>"},
+		{"line break", "here<br>second line"},
+		{"bullet list", `data-fmt="bullet"`},
+		{"bullet items", "<li>alpha</li><li>beta</li></ul>"},
+		{"numbered list", `data-fmt="decimal" data-lvltext="%1."`},
+		{"numbered items", "<li>one</li><li>two</li></ol>"},
+		{"blockquote rule", "border-left:2.25pt solid #c3c7cc"},
 		{"quote text", "quoted wisdom"},
 		{"quote text", "quoted wisdom"},
-		{"code text", "code line 1"},
-		{"table cell", "<td>a</td>"},
-		{"table header bold", "<b>H1</b>"},
+		{"code text", "code line 1<br>code line 2"},
+		{"editor table keeps its spacing", `<table class="of-table" style="width:346.4pt;table-layout:fixed;">`},
+		{"table cell", "<p>a</p></td>"},
+		{"table header bold, centred", `<p style="text-align:center;font-weight:700;">H1</p>`},
+		{"rule", "<hr>"},
 		{"image", `<img src="data:image/png;base64,`},
 		{"image", `<img src="data:image/png;base64,`},
-		{"image width", "width:200px"},
-		{"end text", "end"},
+		{"image size", "width:150pt;height:75pt;"},
+		{"end text", "<p>end</p>"},
 	}
 	}
 	for _, c := range checks {
 	for _, c := range checks {
 		if !strings.Contains(h, c.want) {
 		if !strings.Contains(h, c.want) {
@@ -142,6 +146,14 @@ func TestDocxRoundtrip(t *testing.T) {
 	if !got.PageNumbers {
 	if !got.PageNumbers {
 		t.Errorf("pageNumbers flag lost")
 		t.Errorf("pageNumbers flag lost")
 	}
 	}
+	// the page number the export adds comes back as the switch, not text
+	if strings.Contains(got.FooterHTML, "doc-field") || strings.Contains(got.Footer, "1") {
+		t.Errorf("automatic page number imported as footer content: %q / %q", got.Footer, got.FooterHTML)
+	}
+	// a table in a two-column page is one column wide
+	if i := strings.Index(h, "<table"); i < 0 || !strings.Contains(h[i:], "width:346.4pt;table-layout:fixed") {
+		t.Errorf("table not sized to the column: %.600s", h)
+	}
 }
 }
 
 
 func TestParseDocumentJSON(t *testing.T) {
 func TestParseDocumentJSON(t *testing.T) {

File diff suppressed because it is too large
+ 1897 - 623
src/mod/office/docx_writer.go


+ 208 - 0
src/mod/office/html_helpers.go

@@ -0,0 +1,208 @@
+package office
+
+/*
+	html_helpers.go - small helpers for reading the editors' HTML, shared
+	by the docx, odt and pdf writers.
+*/
+
+import (
+	"fmt"
+	"strconv"
+	"strings"
+
+	"golang.org/x/net/html"
+)
+
+var blockTags = map[string]bool{
+	"p": true, "div": true, "h1": true, "h2": true, "h3": true, "h4": true,
+	"h5": true, "h6": true, "ul": true, "ol": true, "table": true,
+	"blockquote": true, "pre": true, "hr": true, "li": true,
+}
+
+func hasBlockChild(n *html.Node) bool {
+	for c := n.FirstChild; c != nil; c = c.NextSibling {
+		if c.Type == html.ElementNode && blockTags[c.Data] {
+			return true
+		}
+	}
+	return false
+}
+
+func findHTMLNode(n *html.Node, tag string) *html.Node {
+	if n.Type == html.ElementNode && n.Data == tag {
+		return n
+	}
+	for c := n.FirstChild; c != nil; c = c.NextSibling {
+		if f := findHTMLNode(c, tag); f != nil {
+			return f
+		}
+	}
+	return nil
+}
+
+func htmlAttr(n *html.Node, name string) string {
+	for _, a := range n.Attr {
+		if a.Key == name {
+			return a.Val
+		}
+	}
+	return ""
+}
+
+// styleProp extracts one property from an inline style attribute
+func styleProp(style, prop string) string {
+	for _, decl := range strings.Split(style, ";") {
+		kv := strings.SplitN(decl, ":", 2)
+		if len(kv) == 2 && strings.TrimSpace(strings.ToLower(kv[0])) == prop {
+			return strings.TrimSpace(kv[1])
+		}
+	}
+	return ""
+}
+
+// tableColPercents reads the editor's <colgroup><col> widths (percent OR
+// pixel/point units - the column resizer writes px), normalized to 100;
+// equal split when absent or malformed
+func tableColPercents(tbl *html.Node, cols int) []float64 {
+	out := make([]float64, cols)
+	got := 0
+	for cg := tbl.FirstChild; cg != nil; cg = cg.NextSibling {
+		if cg.Type != html.ElementNode || cg.Data != "colgroup" {
+			continue
+		}
+		for col := cg.FirstChild; col != nil && got < cols; col = col.NextSibling {
+			if col.Type != html.ElementNode || col.Data != "col" {
+				continue
+			}
+			ws := strings.TrimSpace(styleProp(htmlAttr(col, "style"), "width"))
+			num := strings.TrimSuffix(strings.TrimSuffix(strings.TrimSuffix(ws, "%"), "px"), "pt")
+			if (strings.HasSuffix(ws, "%") || strings.HasSuffix(ws, "px") || strings.HasSuffix(ws, "pt")) && num != ws {
+				if v, err := strconv.ParseFloat(num, 64); err == nil && v > 0 {
+					out[got] = v // any unit: normalized by the sum below
+					got++
+					continue
+				}
+			}
+			got = 0 // one bad entry: fall back to the equal split
+			break
+		}
+		break
+	}
+	if got != cols {
+		for i := range out {
+			out[i] = 100.0 / float64(cols)
+		}
+		return out
+	}
+	sum := 0.0
+	for _, v := range out {
+		sum += v
+	}
+	if sum > 0 {
+		for i := range out {
+			out[i] = out[i] * 100 / sum
+		}
+	}
+	return out
+}
+
+// tableWidthPct reads the table's own inline width (px, pt or percent) as
+// a percentage of the text width; 100 when absent
+func tableWidthPct(tbl *html.Node) float64 {
+	const textWpx = 620.0
+	ws := strings.TrimSpace(styleProp(htmlAttr(tbl, "style"), "width"))
+	if strings.HasSuffix(ws, "%") {
+		if v, err := strconv.ParseFloat(strings.TrimSuffix(ws, "%"), 64); err == nil && v > 1 {
+			if v > 100 {
+				v = 100
+			}
+			return v
+		}
+	}
+	px := 0.0
+	if strings.HasSuffix(ws, "px") {
+		px, _ = strconv.ParseFloat(strings.TrimSuffix(ws, "px"), 64)
+	} else if strings.HasSuffix(ws, "pt") {
+		pt, _ := strconv.ParseFloat(strings.TrimSuffix(ws, "pt"), 64)
+		px = pt / 0.75
+	}
+	if px > 10 {
+		pct := px * 100 / textWpx
+		if pct > 100 {
+			pct = 100
+		}
+		return pct
+	}
+	return 100
+}
+
+// cssColorHex normalizes "#rgb", "#rrggbb" or "rgb(r, g, b)" to "RRGGBB"
+// ("" when unparseable or transparent)
+func cssColorHex(c string) string {
+	c = strings.TrimSpace(c)
+	if c == "" || c == "transparent" {
+		return ""
+	}
+	if strings.HasPrefix(c, "#") {
+		return hexColor(c, "")
+	}
+	if strings.HasPrefix(c, "rgb") {
+		open := strings.Index(c, "(")
+		close := strings.Index(c, ")")
+		if open < 0 || close <= open {
+			return ""
+		}
+		parts := strings.Split(c[open+1:close], ",")
+		if len(parts) < 3 {
+			return ""
+		}
+		if len(parts) >= 4 {
+			if a, err := strconv.ParseFloat(strings.TrimSpace(parts[3]), 64); err == nil && a == 0 {
+				return ""
+			}
+		}
+		out := ""
+		for i := 0; i < 3; i++ {
+			v, err := strconv.Atoi(strings.TrimSpace(parts[i]))
+			if err != nil || v < 0 || v > 255 {
+				return ""
+			}
+			out += fmt.Sprintf("%02X", v)
+		}
+		return out
+	}
+	switch strings.ToLower(c) {
+	case "black":
+		return "000000"
+	case "white":
+		return "FFFFFF"
+	case "red":
+		return "FF0000"
+	case "blue":
+		return "0000FF"
+	case "green":
+		return "008000"
+	case "gray", "grey":
+		return "808080"
+	}
+	return ""
+}
+
+func textContent(n *html.Node) string {
+	var sb strings.Builder
+	var walk func(*html.Node)
+	walk = func(x *html.Node) {
+		if x.Type == html.TextNode {
+			sb.WriteString(x.Data)
+			return
+		}
+		if x.Type == html.ElementNode && x.Data == "br" {
+			sb.WriteString("\n")
+		}
+		for c := x.FirstChild; c != nil; c = c.NextSibling {
+			walk(c)
+		}
+	}
+	walk(n)
+	return sb.String()
+}

+ 161 - 28
src/web/Office/README.md

@@ -95,8 +95,8 @@ have written. `generate.go -wasm` builds and ships that module, and
 Two capability questions, and they are **not** the same:
 Two capability questions, and they are **not** the same:
 
 
 - `OfficePlatform.hasBackend()` — is there a server? (storage, AGI scripts,
 - `OfficePlatform.hasBackend()` — is there a server? (storage, AGI scripts,
-  the Docs/Sheets PDF renderers — Slides renders its own PDF in the browser
-  and needs no backend for it)
+  the Sheets PDF renderer — Docs and Slides render their own PDF in the
+  browser and need no backend for it)
 - `OfficePlatform.canConvert()` — can this build convert Office formats?
 - `OfficePlatform.canConvert()` — can this build convert Office formats?
 
 
 **Gate anything new on the right one** (details in `CONTRACT.md`), or it will
 **Gate anything new on the right one** (details in `CONTRACT.md`), or it will
@@ -115,12 +115,40 @@ both in a FloatWindow and a plain tab).
 Go structs are the source of truth — they mirror the JS exactly:
 Go structs are the source of truth — they mirror the JS exactly:
 
 
 - **Docs** (`document`): [`docx.go`](../../mod/office/docx.go) —
 - **Docs** (`document`): [`docx.go`](../../mod/office/docx.go) —
-  `{html, page{size, orientation, margins(mm), columns, colGap}, header,
-  footer, hfMode, pageNumbers, comments, trackChanges}`. `html` is a
-  sanitized contenteditable subset (see `sanitizeHtml` in `docs.js`).
-  `hfMode` (`all` | `except-first` | `none`, Format > Header & footer)
-  says which pages repeat the header/footer text; empty means `all`, so
-  documents written before the setting existed keep their behaviour.
+  `{html, page{size, orientation, margins(mm), columns, colGap, headerDist,
+  footerDist}, header, footer, headerHtml, footerHtml, footnotes[{id, html}],
+  lineSpacing, hfMode, pageNumbers, comments, trackChanges}`. `html` is a
+  sanitized contenteditable subset (see `sanitizeHtml` in `docs.js`) — the
+  **rich model** below. `header`/`footer` are the plain-text pair the editor
+  types into; `headerHtml`/`footerHtml` win when set (an imported header with
+  its own typography, a picture, a PAGE field). `lineSpacing` is the
+  document's default multiple (1.15 when absent). `hfMode` (`all` |
+  `except-first` | `none`, Format > Header & footer) says which pages repeat
+  the header/footer; empty means `all`, so documents written before the
+  setting existed keep their behaviour.
+
+  The rich model is plain HTML with inline CSS **in points** plus a few data
+  attributes, so one representation is shared by the editor, the layout
+  engine, the PDF exporter and the DOCX reader/writer:
+  - blocks: `padding-top` = spacing before (a heading's is `margin-top`,
+    which collapses like Google Docs does), `margin-bottom` = after,
+    `margin-left/right` + `text-indent` = indents, `border-*` + `padding-*`
+    = paragraph rules and their space; `data-ls` (multiple),
+    `data-lsexact` / `data-lsmin` (pt), `data-keep-next`,
+    `data-keep-lines`, `data-widow="0"`, `data-page-break-before`,
+    `data-tabs="right:451.3:dot;…"`.
+  - lists: `ol/ul.doc-list[data-num][data-fmt][data-lvltext]` with
+    `padding-left` and `--doc-hang`; the marker text is computed into
+    `li[data-marker]` by the layout engine (Word numbering continues across
+    lists that share a `data-num`).
+  - inline: `span.doc-tab` (a real tab), `sup.doc-fnref[data-fn]`,
+    `span.doc-field[data-field=PAGE|NUMPAGES]`.
+  - pictures: `width/height` in pt, `object-view-box: inset(…)` for a crop,
+    `img.doc-anchor` for one anchored above/below the text.
+  - tables: `table.of-table` with a pt `width`, `table-layout: fixed` and a
+    pt `<colgroup>`; cells state borders/padding/background inline.
+    `data-docx="1"` marks a table laid out the Word way (no spacing around
+    it) — one made in the editor has none and keeps docs.css's 8pt.
 - **Sheets** (`spreadsheet`): [`xlsx.go`](../../mod/office/xlsx.go) —
 - **Sheets** (`spreadsheet`): [`xlsx.go`](../../mod/office/xlsx.go) —
   `{sheets[{name, cells{"A1":{v,s,n}}, colW, rowH, merges, freeze,
   `{sheets[{name, cells{"A1":{v,s,n}}, colW, rowH, merges, freeze,
   charts, filter, cf}], active}`. Cell `v` is the raw input (`=`-prefix =
   charts, filter, cf}], active}`. Cell `v` is the raw input (`=`-prefix =
@@ -160,13 +188,76 @@ body before posting:
 - **Sheets PDF print model** → client sends formatted display strings +
 - **Sheets PDF print model** → client sends formatted display strings +
   styles (`Core.buildPrintModel()` in `sheets.js`) because formula
   styles (`Core.buildPrintModel()` in `sheets.js`) because formula
   evaluation and number formatting live in the client.
   evaluation and number formatting live in the client.
-- **Emoji in Docs PDF** → client rasterizes each emoji to a small PNG
-  (`rasterizeEmojiForPdf` in `docs.js`) because PDF core fonts are
-  Latin-1 and have no emoji glyphs.
+
+### Docs: one layout, drawn three ways
+
+A Docs page looks the same in the editor, in its PDF and (as far as Word's
+model allows) in its `.docx`, because there is only one layout:
+
+- [`docs/docs_layout.js`](docs/docs_layout.js) (`DocsLayout`) lays the live
+  editor DOM out the way a word processor does and paginates it: line
+  heights from real font metrics (`fontRatios`, measured on a 2048px canvas
+  — CSS `line-height: normal` differs per font and per platform), list
+  numbering, tab stops with leaders, table rules compensated for pixel
+  snapping, keep-with-next / keep-lines / widow and orphan control,
+  footnote space at the foot of each page. **A page boundary is real in the
+  DOM**: whatever crosses it is split into two elements — a paragraph at a
+  line, the list or quote around it, a table row cell by cell (a copy of
+  the row takes the rest of every cell) — and a `.doc-autobreak` spacer
+  between the halves carries the second one to the next sheet. Each half is
+  a box of its own, so borders, shading and cell rules end at the bottom of
+  their page. The halves are paired by a token (`data-split` on the head,
+  `data-split-of` on the tail, `data-pair` on the spacer; CSS
+  `.doc-split-head/-tail` drops the spacing and rule at the cut and the
+  marker of a continued list item). Undoing a split moves the tail's content
+  back and rejoins the divided text node, following the caret through it.
+  Editing around a cut goes through the same undo: `unsplitAtCaret` runs
+  before Backspace/Delete at the edge of a cut and before any key typed over
+  a selection that spans one, `unsplitWithin` before a table gains or loses
+  a row or column, and `repairSplits` drops what editing left behind (a tail
+  whose spacer was deleted, a marker copied by Enter). A relayout after
+  typing starts from the page that was edited. Nothing it adds is saved:
+  `stripLayoutArtifacts()` in `docs.js` undoes every split and removes every
+  spacer and computed attribute before a body is serialized (the paste
+  sanitizer strips them too).
+- The paper is **one `.doc-sheet` per page** in `#pageSheets`, behind the
+  transparent `#page`. The gap between two sheets is empty space, not a
+  band painted over one long sheet, and since nothing in the flow sits
+  there, nothing can show through. (A multi-column document is not
+  paginated and keeps a single sheet.) The status bar's "Page N of M"
+  follows the caret, or the middle of the view after a scroll.
+- [`docs/docs_pdf.js`](docs/docs_pdf.js) (`DocsPdf`) reads each sheet of that
+  DOM: text runs at the browser's baselines, fills and borders (a collapsed
+  table rule at the width the document states), pictures with their crop,
+  list markers (a disc/ring/square bullet becomes the shape at the glyph's
+  measured ink box — the only face shipped with the glyph is a CJK one,
+  twice the size), tab leaders and the footnote rule. **The export does not
+  hold the editor.** The page is only held while it is measured (a few
+  hundred ms for 90 pages, in its export state); what to draw is written
+  down as a plain-data display list, and
+  [`common/pdfworker.js`](common/pdfworker.js) assembles the file with
+  [`common/pdfdraw.js`](common/pdfdraw.js) in a Web Worker — embedding and
+  deflating pictures, subsetting fonts and serializing is where the time
+  goes (on the 92-page reference it held the page for ~14s before). The
+  same `pdfdraw.js` runs in the page when a worker cannot be started.
+  Progress shows in `OfficeApp.showProgress`, as in Slides, and the
+  document can be edited meanwhile without changing the file that comes
+  out. Font resolution against the shipped Noto faces, text runs and the
+  raster fallback live in [`common/pdfcore.js`](common/pdfcore.js), also
+  used by Slides (whose exporter still assembles in the page). `docs/backend/docx.agi`'s `export-pdf` and
+  `mod/office/pdf_doc.go` remain for AGI callers (`office.documentToPdf`),
+  but the editor no longer uses them.
+- The DOCX reader/writer (next section) map that same model to
+  WordprocessingML and back.
+
+Check changes against real documents: the round trip docx → editor → PDF
+was tuned against Google Docs' own PDF exports; comparing text line
+positions page by page (PyMuPDF on the PDF, `getClientRects()` on the DOM)
+finds a regression in minutes where eyeballing takes hours.
 
 
 ### Slides: PDF export is rendered in the browser
 ### Slides: PDF export is rendered in the browser
 
 
-**Docs and Sheets export PDF on the server; Slides does not.** Its
+**Sheets exports PDF on the server; Docs and Slides do not.** Slides'
 exporter is [`slides/slides_pdf.js`](slides/slides_pdf.js) (`SlidesPdf`),
 exporter is [`slides/slides_pdf.js`](slides/slides_pdf.js) (`SlidesPdf`),
 built on the vendored `pdf-lib`, and it runs against the very DOM the
 built on the vendored `pdf-lib`, and it runs against the very DOM the
 editor is showing.
 editor is showing.
@@ -591,13 +682,46 @@ the path that honours every mode exactly.
   falls back to the CSS stack. That is the one remaining reason an imported
   falls back to the CSS stack. That is the one remaining reason an imported
   deck can differ visibly from its source: the glyphs are a substitute, so
   deck can differ visibly from its source: the glyphs are a substitute, so
   a line may wrap a word earlier.
   a line may wrap a word earlier.
-- **DOCX pagination** ([`docx_writer.go`](../../mod/office/docx_writer.go)):
-  Word substitutes its own Normal-style defaults (Calibri etc.) unless the
-  style sheet pins the editor's typography into `docDefaults` +
-  `pPrDefault` *and* every named style. That's why `docxStyles` spells out
-  Arial 11pt / 1.5 line-height / explicit spacing everywhere. Change the
-  editor's typography → change it there too, or exported page breaks
-  drift from the editor's.
+- **DOCX import is inheritance too**
+  ([`docx_reader.go`](../../mod/office/docx_reader.go),
+  [`docx_props.go`](../../mod/office/docx_props.go),
+  [`docx_numbering.go`](../../mod/office/docx_numbering.go)): a
+  paragraph's look is docDefaults → the paragraph style's `basedOn` chain
+  → direct `pPr` → character style → run `rPr`. Toggles are tri-state
+  (`<w:b w:val="0"/>` switches bold *off* against a bold style — reading it
+  as "present = on" is the classic bug). Numbering resolves `num` →
+  `abstractNum` → `lvlOverride`, with counters per list id.
+- **Google Docs exports get Google Docs' layout rules** (detected by the
+  all-zero rsids): the empty paragraph above a table loses its spacing
+  after, spacing-before after a page break is dropped, a heading's
+  spacing-before collapses with the spacing-after above it, a picture has
+  1.5pt either side, rows round up to whole pixels, and "Arial Unicode MS"
+  is Arial. Each rule was measured against its own PDF; they are `cv.gdocs`
+  branches so a Word document is not bent by them.
+- **A turned or mirrored picture is baked into the bitmap on import**
+  ([`docx_picture.go`](../../mod/office/docx_picture.go)): a CSS transform
+  would not move the text around it, so the frame is swapped and the crop
+  turned with it, and the picture lays out, prints and saves as it looks.
+- **DOCX export writes what the editor draws**
+  ([`docx_writer.go`](../../mod/office/docx_writer.go)): `docDefaults`,
+  heading styles and `editorBlockCSS` mirror `docs.css` (a `blockquote`'s
+  3px rule and padding, a `pre`'s frame and shading, a `th` centred), so
+  the export starts from the editor's defaults and lays the element's own
+  inline style over them. Paragraph borders use Word's geometry: the text
+  keeps its indent and the rule is drawn `w:space` outside it — the reader
+  maps that back to `margin + border + padding`. A block inside an indent
+  container (the browser's indent command wraps paragraphs in a
+  `blockquote style="margin: 0 0 0 40px"`) takes the container's indent.
+  Three custom styles mark what Word cannot express, so an import restores
+  it exactly: `ArozPageNumber` (the page number the editor draws by itself —
+  it comes back as `pageNumbers`, not footer text), `ArozHorizontalRule`
+  (an `<hr>`) and `ArozEditorTable` (an editor-made table, which keeps its
+  CSS spacing). The editor's plain 9pt grey header/footer also comes back
+  plain. A table or picture in a multi-column page is sized to one column.
+- **The editor model must round-trip.** `TestDocxRichRoundTrip` pins
+  model → docx → model for each construct; when adding one, add a row. A
+  mismatch shows up as layout drift the next time the file is opened, not
+  as an error.
 - **PPTX video/audio are NOT embedded**
 - **PPTX video/audio are NOT embedded**
   ([`pptx_writer.go`](../../mod/office/pptx_writer.go)): embedded media
   ([`pptx_writer.go`](../../mod/office/pptx_writer.go)): embedded media
   (`a:videoFile` + `p14:media` + timing tree, python-pptx-identical
   (`a:videoFile` + `p14:media` + timing tree, python-pptx-identical
@@ -719,13 +843,19 @@ sh ../scripts/check-conventions.sh --diff origin/master
 
 
 ## Ideas / known gaps (future work)
 ## Ideas / known gaps (future work)
 
 
-- **CJK text as text in the Docs and Sheets PDF export.** Slides solved
+- **CJK text as text in the Sheets PDF export.** Slides and Docs solved
   this by shipping the fonts and embedding them in the browser
   this by shipping the fonts and embedding them in the browser
-  (`common/fonts`, `slides_pdf.js`); the Go renderer behind Docs and
-  Sheets still transliterates, because `fpdf`'s core fonts are cp1252.
-  The fix is either to give those two the same browser-side treatment —
-  which is the smaller job, since the machinery now exists — or to teach
-  `mod/office/pdf.go` to embed a CID font.
+  (`common/fonts`, `common/pdfcore.js`); the Go renderer behind Sheets
+  still transliterates, because `fpdf`'s core fonts are cp1252.
+- **Docs line breaking differs from Google Docs in one respect**: Google
+  Docs breaks only at spaces (a word longer than the line is cut at the
+  character), while Chrome also breaks after a hyphen or between quote
+  marks. A long code line can therefore wrap one word differently, which
+  moves the rest of that page by a line. CSS has no switch to take break
+  opportunities away; fixing it means marking them in the text.
+- A DOCX table with no rows (Google Docs writes these) takes no space in
+  the editor; Google Docs gives the heading after it a little less
+  spacing-before.
 - **MicroType Express decompression** so Google-Slides-embedded fonts can
 - **MicroType Express decompression** so Google-Slides-embedded fonts can
   be used (see the format notes) — the last visible gap between an
   be used (see the format notes) — the last visible gap between an
   imported deck and its source.
   imported deck and its source.
@@ -743,9 +873,12 @@ sh ../scripts/check-conventions.sh --diff origin/master
 - Slides: SmartArt (`dgm:`), 3-D effects, shadows and animations are
 - Slides: SmartArt (`dgm:`), 3-D effects, shadows and animations are
   skipped rather than approximated.
   skipped rather than approximated.
 - Real-time collaboration (the `sharedspace` AGI lib was built for this).
 - Real-time collaboration (the `sharedspace` AGI lib was built for this).
-- Docs: footnotes, section breaks, multi-column export to docx/pdf
-  (`page.columns` renders in-editor and exports to docx, but the PDF
-  renderer ignores it).
+- Docs: section breaks (one page setup per document), and **pagination of
+  multi-column documents** — `page.columns` renders as CSS columns with
+  dotted page guides only, so text runs across the sheet boundaries in the
+  page view and in the browser PDF (which draws that view); the `.docx`
+  export writes real Word columns. Paginating columns means giving
+  `DocsLayout.paginate` a column-balancing pass.
 - Sheets PDF: merged-cell rendering in the print model.
 - Sheets PDF: merged-cell rendering in the print model.
 - Slides: shape text with per-run styling in pptx (currently
 - Slides: shape text with per-run styling in pptx (currently
   object-level bold/italic/color only).
   object-level bold/italic/color only).

+ 11 - 4
src/web/Office/common/CONTRACT.md

@@ -58,7 +58,8 @@ Apps are registered in `Office/init.agi` (already done — do not edit it).
     <script src="../common/colorpicker.js"></script>
     <script src="../common/colorpicker.js"></script>
     <script src="../common/clipboard.js"></script>
     <script src="../common/clipboard.js"></script>
     <!-- optional: ../common/charts.js, ../common/textedit.js,
     <!-- optional: ../common/charts.js, ../common/textedit.js,
-         ../common/lib/marked.min.js, ../common/lib/pdf-lib.min.js,
+         ../common/lib/marked.min.js, ../common/lib/pdf-lib.min.js +
+         ../common/pdfcore.js,
          ../common/lib/html2canvas.min.js -->
          ../common/lib/html2canvas.min.js -->
     <!-- with fonts.js: ../common/fonts/fonts.css declares the shipped
     <!-- with fonts.js: ../common/fonts/fonts.css declares the shipped
          document faces; an app that lets the user pick a font needs it -->
          document faces; an app that lets the user pick a font needs it -->
@@ -714,12 +715,18 @@ downstream has to know about them.
 
 
 - `marked.min.js` — Markdown → HTML (Docs import)
 - `marked.min.js` — Markdown → HTML (Docs import)
 - `pdf-lib.min.js` — PDF generation (global `PDFLib`). Used by the Slides
 - `pdf-lib.min.js` — PDF generation (global `PDFLib`). Used by the Slides
-  PDF export, which builds the file in the browser out of real PDF objects
-  (`slides/slides_pdf.js`).
+  and Docs PDF exports, which build the file in the browser out of real PDF
+  objects (`slides/slides_pdf.js`, `docs/docs_pdf.js`) on the shared
+  `common/pdfcore.js` (`OfficePdfCore`: font resolution against the shipped
+  faces, text runs, clip paths, raster fallback) — load it after pdf-lib.
+  Docs hands the actual PDF assembly to a Web Worker: `common/pdfworker.js`
+  (which `importScripts` pdf-lib, fontkit and `common/pdfdraw.js`) turns a
+  plain-data display list into the file; `pdfdraw.js` is also loaded in the
+  page as the fallback when no worker can start.
 - `fontkit.umd.min.js` — `@pdf-lib/fontkit`, which is what lets `pdf-lib`
 - `fontkit.umd.min.js` — `@pdf-lib/fontkit`, which is what lets `pdf-lib`
   embed a font of our own. **Loaded on demand, not from the page**: it is
   embed a font of our own. **Loaded on demand, not from the page**: it is
   the largest script here and only an export needs it (`loadFontkit` in
   the largest script here and only an export needs it (`loadFontkit` in
-  `slides_pdf.js` injects the tag). Its subsetter has sharp edges that the
+  `common/pdfcore.js` injects the tag). Its subsetter has sharp edges that the
   shipped fonts are built to avoid — see `fonts/README.md`.
   shipped fonts are built to avoid — see `fonts/README.md`.
 - `html2canvas.min.js` — DOM → canvas (Slides PNG export). **Not** the
 - `html2canvas.min.js` — DOM → canvas (Slides PNG export). **Not** the
   first choice for the PDF export's raster fallback: it re-implements
   first choice for the PDF export's raster fallback: it re-implements

+ 804 - 0
src/web/Office/common/pdfcore.js

@@ -0,0 +1,804 @@
+/*
+    ArozOS Office - PDF core
+    ========================
+    The parts of a browser-side PDF exporter that do not care what kind of
+    document is being drawn, shared by Slides (slides/slides_pdf.js) and
+    Docs (docs/docs_pdf.js). Both draw the very DOM their editor shows, so
+    both need the same answers to the same questions: which font a PDF can
+    show a character in, where the browser put a line of text and its
+    baseline, how a picture gets embedded once, and what to do with the one
+    element nothing else can express.
+
+    The notes on each part were written for Slides and still hold word for
+    word for Docs; "slide" there reads as "page".
+
+    Requires pdf-lib (PDFLib) and OfficeFonts (common/fonts.js); fontkit is
+    fetched on the first export.
+*/
+
+var OfficePdfCore = (function () {
+    "use strict";
+
+    var PX_TO_PT = 0.75;
+    var RASTER_SCALE = 3;      // device pixels per css px for a fallback raster
+
+    /* ---------------- fonts ---------------- */
+
+    // where fontkit lives; it is fetched only when an export runs
+    var FONTKIT_URL = "../common/lib/fontkit.umd.min.js";
+
+    /* Families that are metric-compatible with a standard PDF font, so
+       text set in them lands in exactly the same place as on screen. */
+    var STD_FAMILIES = {
+        "helvetica": "Helvetica", "arial": "Helvetica", "liberation sans": "Helvetica",
+        "arimo": "Helvetica", "sans-serif": "Helvetica", "nimbus sans": "Helvetica",
+        "times": "TimesRoman", "times new roman": "TimesRoman", "serif": "TimesRoman",
+        "liberation serif": "TimesRoman", "tinos": "TimesRoman", "nimbus roman": "TimesRoman",
+        "courier": "Courier", "courier new": "Courier", "monospace": "Courier",
+        "liberation mono": "Courier", "cousine": "Courier", "nimbus mono": "Courier"
+    };
+    var STD_VARIANTS = {
+        Helvetica: ["Helvetica", "HelveticaBold", "HelveticaOblique", "HelveticaBoldOblique"],
+        TimesRoman: ["TimesRoman", "TimesRomanBold", "TimesRomanItalic", "TimesRomanBoldItalic"],
+        Courier: ["Courier", "CourierBold", "CourierOblique", "CourierBoldOblique"]
+    };
+    var GENERICS = { "sans-serif": 1, "serif": 1, "monospace": 1, "cursive": 1, "fantasy": 1 };
+
+    /* WinAnsi is what the standard fonts can encode. A handful of code
+       points above U+00FF are in it too, but keeping to the Latin-1 range
+       is the rule that can be checked without a table. */
+    function winAnsiCp(cp) {
+        return cp >= 32 && cp <= 255;
+    }
+    /* haveFamily reports whether a family is actually installed. It has to
+       be measured: document.fonts.check() answers "is it loaded", and for a
+       local family Chrome says yes whatever name you give it. The reliable
+       test is the old one - render a probe string backed by two different
+       generics; a family that exists overrides both and comes out a
+       different width from each, one that does not falls through to the
+       generic and matches it exactly.
+
+       This matters because a deck may ask for a font the machine does not
+       have: the browser laid the text out in the fallback, so the fallback
+       is what the PDF must match - and that may well be one it can show. */
+    var PROBE = "mmmmmmmmwwwwwwwwiiiiiiiil1I0Oo";
+    var familyKnown = {};
+    var probeCtx = null;
+    var probeBase = null;
+    function haveFamily(name) {
+        if (familyKnown[name] !== undefined) return familyKnown[name];
+        try {
+            if (!probeCtx) {
+                probeCtx = document.createElement("canvas").getContext("2d");
+                probeBase = {};
+                ["monospace", "serif"].forEach(function (g) {
+                    probeCtx.font = '72px ' + g;
+                    probeBase[g] = probeCtx.measureText(PROBE).width;
+                });
+            }
+            var found = true;
+            ["monospace", "serif"].forEach(function (g) {
+                probeCtx.font = '72px "' + name.replace(/"/g, "") + '", ' + g;
+                if (probeCtx.measureText(PROBE).width === probeBase[g]) found = false;
+            });
+            familyKnown[name] = found;
+        } catch (e) {
+            familyKnown[name] = true;
+        }
+        return familyKnown[name];
+    }
+
+    /* familyList turns a computed font-family into the list the browser
+       walks, with the shipped faces on the end. The tail matters for old
+       content: a <font face="..."> names one family and nothing else, and
+       without it a character that family has no glyph for would have
+       nowhere to go. */
+    function familyList(cssFamily) {
+        var out = [], seen = {};
+        function add(n) {
+            n = String(n).trim().replace(/^["']|["']$/g, "");
+            if (!n || seen[n.toLowerCase()]) return;
+            seen[n.toLowerCase()] = true;
+            out.push(n);
+        }
+        String(cssFamily || "").split(",").forEach(add);
+        OfficeFonts.FALLBACK.forEach(add);
+        return out;
+    }
+
+    /* fontkit is what lets pdf-lib embed a font file of our own. It is the
+       largest script the app has and only an export needs it, so it is
+       fetched on the first export and not before. */
+    var fontkitPromise = null;
+    function loadFontkit() {
+        if (fontkitPromise) return fontkitPromise;
+        if (window.fontkit) {
+            fontkitPromise = Promise.resolve(window.fontkit);
+            return fontkitPromise;
+        }
+        fontkitPromise = new Promise(function (resolve, reject) {
+            var el = document.createElement("script");
+            el.src = FONTKIT_URL;
+            el.onload = function () {
+                if (window.fontkit) resolve(window.fontkit);
+                else reject(new Error("the font toolkit did not load"));
+            };
+            el.onerror = function () { reject(new Error("the font toolkit did not load")); };
+            document.head.appendChild(el);
+        });
+        return fontkitPromise;
+    }
+
+    /* makeFonts is the document's font supply.
+
+       A standard font is there for the asking. A shipped face goes through
+       two stages, and the split matters:
+
+         want()  fetches the file and parses it, which is what answers "does
+                 this face have a glyph for this character". Asynchronous,
+                 so a slide says what it needs, waits (ready), then draws.
+         use()   puts it in the PDF. Only faces that really get drawn with
+                 may be embedded: the embedder subsets a font down to the
+                 glyphs that were asked of it, and a subset of nothing is
+                 not a font any more - a CFF one fails outright on save.
+
+       Drawing itself stays synchronous, which is what lets a fragment be
+       measured and placed in one pass. */
+    function makeFonts(pdfDoc, fontkit) {
+        var std = {};
+        var faces = {};        // url -> { kit, font? }, or null when it failed
+        var asked = {};        // url -> Promise, set the moment it is wanted
+        var wanted = [];
+
+        function stdFont(family, bold, italic) {
+            var names = STD_VARIANTS[family] || STD_VARIANTS.Helvetica;
+            var key = names[(bold ? 1 : 0) + (italic ? 2 : 0)];
+            if (!std[key]) std[key] = pdfDoc.embedStandardFont(PDFLib.StandardFonts[key]);
+            return std[key];
+        }
+
+        function want(family, bold, italic) {
+            var face = OfficeFonts.faceFor(family, bold, italic);
+            if (!face || asked[face.url]) return;
+            asked[face.url] = fetch(face.url).then(function (r) {
+                if (!r.ok) throw new Error("cannot read " + face.url);
+                return r.arrayBuffer();
+            }).then(function (buf) {
+                var bytes = new Uint8Array(buf);
+                faces[face.url] = { bytes: bytes, kit: fontkit.create(bytes) };
+            }, function () {
+                // a font that will not load is not a reason to fail the
+                // export: the next family in the stack gets the character
+                faces[face.url] = null;
+            });
+            wanted.push(asked[face.url]);
+        }
+
+        function use(family, bold, italic) {
+            var face = OfficeFonts.faceFor(family, bold, italic);
+            if (!face) return;
+            var rec = faces[face.url];
+            if (!rec || rec.font || rec.embedding) return;
+            rec.embedding = pdfDoc.embedFont(rec.bytes, { subset: true })
+                .then(function (font) { rec.font = font; });
+            wanted.push(rec.embedding);
+        }
+
+        function ready() {
+            var all = wanted;
+            wanted = [];
+            if (!all.length) return Promise.resolve();
+            return Promise.all(all).then(function () { });
+        }
+
+        // shipped hands back a loaded face, null while it is not there
+        function shipped(family, bold, italic) {
+            var face = OfficeFonts.faceFor(family, bold, italic);
+            if (!face) return null;
+            var rec = faces[face.url];
+            if (!rec) return null;
+            return {
+                font: rec.font, kit: rec.kit,
+                synthBold: face.synthBold, synthItalic: face.synthItalic
+            };
+        }
+
+        // tried says whether asking again could still change the answer
+        function tried(family, bold, italic) {
+            var face = OfficeFonts.faceFor(family, bold, italic);
+            return !face || !!asked[face.url];
+        }
+
+        return {
+            std: stdFont, want: want, use: use,
+            ready: ready, shipped: shipped, tried: tried
+        };
+    }
+
+    /* resolveChar walks a font stack the way the browser does and says what
+       the PDF can put this one character in:
+
+         { std }      one of the 14 standard fonts
+         { shipped }  a face the suite ships, already embedded
+         { need }     a shipped face that is named but not loaded yet, so
+                      the answer is not known until it is
+         null         nothing here can show this character
+
+       A family that is neither - a system font - is stepped over rather
+       than used: its bytes are unreadable, so the character goes to the
+       next entry, which is the shipped face for its script. */
+    function resolveChar(cp, names, fonts, bold, italic) {
+        for (var i = 0; i < names.length; i++) {
+            var name = names[i];
+            var key = name.toLowerCase();
+            if (OfficeFonts.isShipped(name)) {
+                var rec = fonts.shipped(name, bold, italic);
+                if (!rec) {
+                    if (!fonts.tried(name, bold, italic)) return { need: name };
+                    continue;
+                }
+                if (rec.kit && rec.kit.hasGlyphForCodePoint &&
+                    !rec.kit.hasGlyphForCodePoint(cp)) continue;
+                return { shipped: name };
+            }
+            if (STD_FAMILIES[key] && (GENERICS[key] || haveFamily(name)) && winAnsiCp(cp)) {
+                return { std: STD_FAMILIES[key] };
+            }
+        }
+        return null;
+    }
+
+    /* segmentText cuts a fragment into the pieces that share one font, the
+       way a browser does per character. Returns null when any character has
+       nowhere to go, which is the signal to rasterize instead. */
+    function segmentText(text, names, fonts, bold, italic) {
+        var segs = [], cur = null;
+        for (var i = 0; i < text.length; i++) {
+            var cp = text.codePointAt(i);
+            var ch = String.fromCodePoint(cp);
+            if (ch.length > 1) i++;          // a surrogate pair
+            var res = resolveChar(cp, names, fonts, bold, italic);
+            if (!res || res.need) return null;
+            var key = res.std ? "s:" + res.std : "f:" + res.shipped;
+            if (cur && cur.key === key) cur.text += ch;
+            else { cur = { key: key, res: res, text: ch }; segs.push(cur); }
+        }
+        return segs;
+    }
+
+    function faceOf(res, fonts, bold, italic) {
+        if (res.std) return { font: fonts.std(res.std, bold, italic), synthBold: false };
+        var rec = fonts.shipped(res.shipped, bold, italic);
+        return rec ? { font: rec.font, synthBold: rec.synthBold } : null;
+    }
+
+    /* Where the baseline sits is the browser's decision, and the exporter
+       has to ask rather than compute: a system font's metrics are not
+       readable from the page, and even for a font that is, the numbers the
+       file states are not always the ones the browser uses.
+
+       A canvas answers it. measureText reports the ascent and descent the
+       browser resolved for a font stack, which is exactly what it used to
+       lay the text out - so the two cannot drift apart.
+
+       This is also why the run's own rect is the reference: the rects a
+       Range hands back for text are the content box, ascent plus descent
+       tall, not the line box. The baseline is therefore an ascent below the
+       top of the rect, with the halving below for the case where a browser
+       hands back the taller box instead. */
+    var metricsCache = {};
+    var metricsCtx = null;
+    function fontMetricsOf(cs) {
+        var font = cs.fontStyle + " " + cs.fontWeight + " " + cs.fontSize + " " + cs.fontFamily;
+        if (metricsCache[font]) return metricsCache[font];
+        var size = parseFloat(cs.fontSize) || 12;
+        var m = null;
+        try {
+            if (!metricsCtx) metricsCtx = document.createElement("canvas").getContext("2d");
+            metricsCtx.font = font;
+            var tm = metricsCtx.measureText("Hxg");
+            if (tm && tm.fontBoundingBoxAscent !== undefined) {
+                m = { ascent: tm.fontBoundingBoxAscent, descent: tm.fontBoundingBoxDescent };
+            }
+        } catch (e) { /* fall through to the estimate */ }
+        // a browser without the font bounding box: the usual proportions
+        if (!m) m = { ascent: size * 0.9, descent: size * 0.22 };
+        metricsCache[font] = m;
+        return m;
+    }
+
+    /* eachTextNode is the walk both the font pre-pass and the run collector
+       make, kept in one place so they cannot disagree about what counts as
+       text on the slide. */
+    function eachTextNode(rootEl, fn) {
+        var walker = document.createTreeWalker(rootEl, NodeFilter.SHOW_TEXT, null);
+        var node;
+        while ((node = walker.nextNode())) {
+            var text = node.nodeValue;
+            if (!text || !text.trim()) continue;
+            var parent = node.parentElement;
+            if (!parent) continue;
+            var cs = window.getComputedStyle(parent);
+            if (cs.visibility === "hidden" || cs.display === "none") continue;
+            // source newlines and tabs are whitespace the browser already
+            // collapsed; they must not reach a font, but the string has to
+            // keep its length - the line split indexes back into the node
+            fn(node, text.replace(/[\u0000-\u001F\u007F]/g, " "), cs);
+        }
+    }
+
+    /* ---------------- small helpers ---------------- */
+
+    function px(v) { return v * PX_TO_PT; }
+    function clamp(v, a, b) { return Math.max(a, Math.min(b, v)); }
+
+    /* parseFill returns both halves of a CSS colour: the colour itself and
+       its alpha. The alpha matters - a table's header band is a translucent
+       wash of the theme accent, and dropping it turns a tint into a slab. */
+    function parseFill(css) {
+        if (!css) return null;
+        var m = /^rgba?\(([^)]+)\)$/i.exec(String(css).trim());
+        if (m) {
+            var p = m[1].split(",").map(function (x) { return parseFloat(x); });
+            var a = p.length >= 4 ? clamp(p[3], 0, 1) : 1;
+            if (a === 0) return null;
+            return {
+                c: PDFLib.rgb(clamp(p[0] / 255, 0, 1), clamp(p[1] / 255, 0, 1), clamp(p[2] / 255, 0, 1)),
+                a: a
+            };
+        }
+        var t = String(css).trim();
+        var h = /^#([0-9a-f]{3}|[0-9a-f]{6}|[0-9a-f]{8})$/i.exec(t);
+        if (!h) return null;
+        var v = h[1];
+        var alpha = 1;
+        if (v.length === 8) { alpha = parseInt(v.substring(6, 8), 16) / 255; v = v.substring(0, 6); }
+        if (v.length === 3) v = v[0] + v[0] + v[1] + v[1] + v[2] + v[2];
+        if (alpha === 0) return null;
+        var n = parseInt(v, 16);
+        return {
+            c: PDFLib.rgb(((n >> 16) & 255) / 255, ((n >> 8) & 255) / 255, (n & 255) / 255),
+            a: alpha
+        };
+    }
+    // parseColor is parseFill when only the colour is wanted
+    /* contrastOf answers "what colour shows up on this fill" - only used
+       for a shape's markings when it has no stroke colour of its own */
+    function contrastOf(css) {
+        var f = parseFill(css);
+        if (!f) return "#333333";
+        var lum = 0.299 * f.c.red + 0.587 * f.c.green + 0.114 * f.c.blue;
+        return lum > 0.6 ? "#333333" : "#ffffff";
+    }
+
+    function parseColor(css) {
+        var f = parseFill(css);
+        return f ? f.c : null;
+    }
+
+    function dataUrlBytes(src) {
+        var comma = String(src || "").indexOf(",");
+        if (comma < 0) return null;
+        var head = src.substring(0, comma);
+        if (head.indexOf(";base64") < 0) return null;
+        var bin = atob(src.substring(comma + 1));
+        var out = new Uint8Array(bin.length);
+        for (var i = 0; i < bin.length; i++) out[i] = bin.charCodeAt(i);
+        return { bytes: out, mime: (/^data:([^;]+)/.exec(head) || [])[1] || "" };
+    }
+
+    /* ---------------- the drawing context ---------------- */
+
+    /* Page is a thin wrapper that flips the y axis once: the editor's
+       coordinates run down from the top-left of the slide, a PDF page's run
+       up from the bottom-left, and mixing the two up is the single easiest
+       way to get an export subtly wrong. */
+    function Page(page, pdfDoc, fonts, heightPx) {
+        this.p = page;
+        this.doc = pdfDoc;
+        this.fonts = fonts;
+        this.h = heightPx;
+        this.gsCache = {};
+    }
+    Page.prototype.y = function (topPx) { return px(this.h - topPx); };
+
+    Page.prototype.rect = function (x, y, w, h, opts) {
+        this.p.drawRectangle({
+            x: px(x), y: this.y(y + h), width: px(w), height: px(h),
+            color: opts.fill || undefined,
+            borderColor: opts.stroke || undefined,
+            borderWidth: opts.strokeW ? px(opts.strokeW) : undefined,
+            borderDashArray: opts.dash ? [px(opts.strokeW * 3), px(opts.strokeW * 2)] : undefined,
+            opacity: opts.fillOpacity !== undefined ? opts.fillOpacity : opts.opacity,
+            borderOpacity: opts.strokeOpacity !== undefined ? opts.strokeOpacity : opts.opacity
+        });
+    };
+
+    // ops pushes raw content-stream operators, which is how the clip paths
+    // and the matrices below are expressed
+    Page.prototype.ops = function (list) {
+        this.p.pushOperators.apply(this.p, list);
+    };
+    Page.prototype.save = function () { this.ops([PDFLib.pushGraphicsState()]); };
+    Page.prototype.restore = function () { this.ops([PDFLib.popGraphicsState()]); };
+
+    // alpha returns the name of an ExtGState for a given opacity, making one
+    // only the first time each distinct value is used on this page
+    Page.prototype.alpha = function (a) {
+        var key = "a" + Math.round(a * 1000);
+        if (!this.gsCache[key]) {
+            var ref = this.doc.context.register(this.doc.context.obj({
+                Type: "ExtGState", ca: a, CA: a
+            }));
+            this.p.node.setExtGState(PDFLib.PDFName.of(key), ref);
+            this.gsCache[key] = true;
+        }
+        return key;
+    };
+
+    /* ---------------- text ---------------- */
+
+    /* A run is one uniformly formatted fragment of a single line, measured
+       off the live DOM: its box, its baseline and the style in force. */
+    function collectRuns(rootEl, origin) {
+        var runs = [];
+        eachTextNode(rootEl, function (node, text, cs) {
+            // one entry per line box the fragment occupies
+            var range = document.createRange();
+            range.selectNodeContents(node);
+            var rects = Array.prototype.slice.call(range.getClientRects());
+            if (!rects.length) return;
+            var shared = {
+                origin: origin,
+                names: familyList(cs.fontFamily),
+                metrics: fontMetricsOf(cs),
+                size: parseFloat(cs.fontSize) || 12,
+                weight: parseInt(cs.fontWeight, 10) || (cs.fontWeight === "bold" ? 700 : 400),
+                italic: cs.fontStyle === "italic" || cs.fontStyle === "oblique",
+                underline: cs.textDecorationLine.indexOf("underline") >= 0,
+                strike: cs.textDecorationLine.indexOf("line-through") >= 0,
+                color: cs.color,
+                background: cs.backgroundColor
+            };
+            splitByLine(node, text, rects).forEach(function (ln) {
+                var r = Object.create(shared);
+                r.text = ln.text;
+                r.rect = ln.rect;
+                runs.push(r);
+            });
+        });
+        return runs;
+    }
+
+    /* splitByLine maps a text node's client rects back onto the substrings
+       that produced them, so each line can be drawn at its own baseline.
+       Character-by-character is the only reliable way: the browser decides
+       where the break went, and only it knows. */
+    function splitByLine(node, text, lineRects) {
+        if (lineRects.length === 1) {
+            return [{ text: text, rect: lineRects[0] }];
+        }
+        var range = document.createRange();
+        var out = [];
+        var cur = "";
+        var curTop = null;
+        var curRect = null;
+        for (var i = 0; i < text.length; i++) {
+            range.setStart(node, i);
+            range.setEnd(node, i + 1);
+            var r = range.getBoundingClientRect();
+            if (r.width === 0 && r.height === 0) { cur += text[i]; continue; }
+            var top = Math.round(r.top * 10) / 10;
+            if (curTop === null || Math.abs(top - curTop) < 0.6) {
+                if (curTop === null) { curTop = top; curRect = { left: r.left, top: r.top, bottom: r.bottom, right: r.right }; }
+                else { curRect.right = Math.max(curRect.right, r.right); }
+                cur += text[i];
+            } else {
+                out.push({ text: cur, rect: curRect });
+                cur = text[i];
+                curTop = top;
+                curRect = { left: r.left, top: r.top, bottom: r.bottom, right: r.right };
+            }
+        }
+        if (cur !== "" && curRect) out.push({ text: cur, rect: curRect });
+        return out.length ? out : [{ text: text, rect: lineRects[0] }];
+    }
+
+    // canDrawAsText is the whole fallback decision, in one place: every
+    // character of every run has to have a font that can show it
+    function canDrawAsText(runs, fonts) {
+        for (var i = 0; i < runs.length; i++) {
+            var r = runs[i];
+            if (!segmentText(r.text, r.names, fonts, r.weight >= 600, r.italic)) return false;
+        }
+        return true;
+    }
+
+    /* drawFragment puts one line fragment on the page as real text, in as
+       many pieces as it takes fonts to spell it.
+
+       fitPx, when it is given, is the width the browser gave the fragment.
+       The text is squeezed or stretched to exactly that with Tz, which
+       costs nothing when the PDF font is the one the browser used (the
+       ratio is 1) and is what keeps a substituted face - a system font we
+       could not embed - from pushing the rest of the line out of place.
+
+       Bold that a shipped face does not have is stroked rather than filled,
+       at the width the browser smears it by. Neither side moves the advance
+       widths, so the two stay in step. */
+    function drawFragment(pg, spec) {
+        var fonts = pg.fonts;
+        var segs = segmentText(spec.text, spec.names, fonts, spec.bold, spec.italic);
+        if (!segs || !segs.length) return 0;
+
+        var total = 0;
+        for (var i = 0; i < segs.length; i++) {
+            var face = faceOf(segs[i].res, fonts, spec.bold, spec.italic);
+            if (!face) return 0;
+            segs[i].face = face;
+            segs[i].w = face.font.widthOfTextAtSize(segs[i].text, spec.sizePx);
+            total += segs[i].w;
+        }
+
+        var scale = 1;
+        if (spec.fitPx > 0 && total > 0) {
+            var ratio = spec.fitPx / total;
+            // a ratio far from 1 means the measurement, not the font, is
+            // wrong (a collapsed space, a transform) - leave it alone
+            if (ratio > 0.5 && ratio < 2 && Math.abs(ratio - 1) > 0.005) scale = ratio;
+        }
+
+        var col = spec.color || PDFLib.rgb(0, 0, 0);
+        var cursor = spec.xPx;
+        segs.forEach(function (seg) {
+            pg.save();
+            var ops = [];
+            if (scale !== 1) ops.push(PDFLib.setCharacterSqueeze(scale * 100));
+            if (seg.face.synthBold) {
+                ops.push(PDFLib.setTextRenderingMode(PDFLib.TextRenderingMode.FillAndOutline));
+                ops.push(PDFLib.setLineWidth(px(spec.sizePx / 28)));
+                ops.push(PDFLib.setStrokingColor(col));
+            }
+            if (ops.length) pg.ops(ops);
+            pg.p.drawText(seg.text, {
+                x: px(cursor), y: pg.y(spec.baselinePx),
+                size: px(spec.sizePx), font: seg.face.font, color: col
+            });
+            pg.restore();
+            cursor += seg.w * scale;
+        });
+        return total * scale;
+    }
+
+    /* drawRuns puts every run on the page at the baseline the browser laid
+       it out on, in the width the browser gave it. */
+    function drawRuns(pg, runs) {
+        runs.forEach(function (r) {
+            var x = r.origin.x + (r.rect.left - r.origin.left);
+            var top = r.origin.y + (r.rect.top - r.origin.top);
+            var lineH = r.rect.bottom - r.rect.top;
+            var domW = r.rect.right - r.rect.left;
+            var glyphH = r.metrics.ascent + r.metrics.descent;
+            var baseline = (lineH - glyphH) / 2 + r.metrics.ascent;
+            var col = parseColor(r.color) || PDFLib.rgb(0, 0, 0);
+            var bg = parseFill(r.background);
+            if (bg) pg.rect(x, top, domW, lineH, { fill: bg.c, fillOpacity: bg.a });
+            // a fragment that starts or ends on a space cannot be fitted to
+            // its rect: the browser collapses those, the measurement does not
+            var w = drawFragment(pg, {
+                text: r.text, names: r.names,
+                bold: r.weight >= 600, italic: r.italic,
+                sizePx: r.size, xPx: x, baselinePx: top + baseline,
+                color: col, fitPx: /^\s|\s$/.test(r.text) ? 0 : domW
+            });
+            if (r.underline || r.strike) {
+                var yOff = r.underline ? baseline + r.size * 0.11 : baseline - r.size * 0.28;
+                pg.rect(x, top + yOff, w || domW, Math.max(0.7, r.size * 0.06), { fill: col });
+            }
+        });
+    }
+
+    /* ---------------- rasterizing one element ---------------- */
+
+    /* The documented last resort, and it has one rule: the pixels come out
+       of a render of the whole slide, then get cropped to the element that
+       needed them.
+
+       Rasterizing an element on its own looks tempting and is wrong.
+       html2canvas re-renders a *clone*, and a clone torn out of its
+       absolutely positioned parent loses the width it was laid out in -
+       so the text re-wraps and the export stops matching the editor,
+       which is the one thing it must not do. Rendering the slide keeps
+       every element in the context it was measured in.
+
+       What lands in the PDF is still only that element's own box: one
+       image, at its own position, with everything else on the page a real
+       PDF object. */
+    /* The picture is taken through an SVG <foreignObject>, which means the
+       *browser* lays the element out and paints it - the same engine, the
+       same fonts, the same line breaks as the editor.
+
+       This is not the obvious choice, so: html2canvas was tried first and
+       is wrong for this. It re-implements layout over a clone, and on
+       mixed CJK/Latin text with pre-wrap it breaks lines somewhere else
+       than the browser did, which is precisely the failure this whole
+       rework exists to remove. A foreignObject cannot re-wrap anything,
+       because it is not re-laying anything out.
+
+       The cost is that computed styles have to be inlined onto the clone
+       (an SVG image cannot reach the page's stylesheets) and every
+       resource must already be a data URL - which, in a slide, it is. */
+    function rasterizeElement(el, w, h) {
+        try {
+            var clone = el.cloneNode(true);
+            inlineStyles(el, clone);
+            // the foreignObject supplies the box, so the clone must not
+            // carry the absolute placement it had on the slide
+            clone.style.position = "static";
+            clone.style.left = "auto";
+            clone.style.top = "auto";
+            clone.style.transform = "none";
+            clone.style.margin = "0";
+            clone.style.width = w + "px";
+            clone.style.height = h + "px";
+
+            var cw = Math.max(1, Math.ceil(w)), ch = Math.max(1, Math.ceil(h));
+            /* The markup inside a foreignObject has to be well-formed XML,
+               and innerHTML is not: it writes <br> unclosed, which makes the
+               whole SVG fail to parse and the element vanish from the page.
+               XMLSerializer writes real XHTML, so it cannot. */
+            var wrap = document.createElementNS("http://www.w3.org/1999/xhtml", "div");
+            wrap.setAttribute("style", "width:" + cw + "px;height:" + ch + "px;");
+            wrap.appendChild(clone);
+            var xhtml = new XMLSerializer().serializeToString(wrap);
+            var svg = '<svg xmlns="http://www.w3.org/2000/svg" width="' + cw +
+                '" height="' + ch + '"><foreignObject x="0" y="0" width="' + cw +
+                '" height="' + ch + '">' + xhtml + "</foreignObject></svg>";
+            var url = "data:image/svg+xml;charset=utf-8," + encodeURIComponent(svg);
+            return new Promise(function (resolve) {
+                var img = new Image();
+                img.onload = function () {
+                    try {
+                        var c = document.createElement("canvas");
+                        c.width = Math.max(1, Math.round(cw * RASTER_SCALE));
+                        c.height = Math.max(1, Math.round(ch * RASTER_SCALE));
+                        var g = c.getContext("2d");
+                        g.drawImage(img, 0, 0, c.width, c.height);
+                        resolve(c.toDataURL("image/png"));
+                    } catch (e) { resolve(null); }
+                };
+                img.onerror = function () { resolve(null); };
+                img.src = url;
+            });
+        } catch (e) {
+            return Promise.resolve(null);
+        }
+    }
+
+    /* rasterizeFallback is the very last resort, for the case where even
+       the foreignObject route fails. html2canvas re-implements layout and
+       can break mixed-script lines somewhere the browser did not, so it is
+       only ever reached when the alternative is dropping the element
+       from the page entirely - which would be worse. */
+    function rasterizeFallback(el, w, h) {
+        if (typeof html2canvas === "undefined") return Promise.resolve(null);
+        return html2canvas(el, {
+            scale: RASTER_SCALE, useCORS: true, backgroundColor: null, logging: false
+        }).then(function (canvas) {
+            return canvas.toDataURL("image/png");
+        }).catch(function () { return null; });
+    }
+
+    /* inlineStyles copies the computed style of every node in the subtree
+       onto the clone, because the SVG image has no access to the page's
+       stylesheets. Slide markup is small - a few divs and spans - so
+       walking the whole property list is affordable and leaves nothing out. */
+    function inlineStyles(src, dst) {
+        var cs = window.getComputedStyle(src);
+        var out = "";
+        for (var i = 0; i < cs.length; i++) {
+            var prop = cs[i];
+            out += prop + ":" + cs.getPropertyValue(prop) + ";";
+        }
+        dst.setAttribute("style", out);
+        var a = src.children, b = dst.children;
+        for (var j = 0; j < a.length && j < b.length; j++) inlineStyles(a[j], b[j]);
+    }
+
+    /* ---------------- images ---------------- */
+
+    /* filteredImageData re-encodes a picture through a canvas when it
+       carries a colour treatment. Applying a filter is a pixel operation in
+       any renderer, so this is the correct way to do it, not a fallback. */
+    function filteredImageData(src, filter, naturalW, naturalH) {
+        return new Promise(function (resolve) {
+            var img = new Image();
+            img.onload = function () {
+                try {
+                    var c = document.createElement("canvas");
+                    c.width = naturalW || img.naturalWidth;
+                    c.height = naturalH || img.naturalHeight;
+                    var g = c.getContext("2d");
+                    g.filter = filter;
+                    g.drawImage(img, 0, 0, c.width, c.height);
+                    resolve(c.toDataURL("image/png"));
+                } catch (e) { resolve(null); }
+            };
+            img.onerror = function () { resolve(null); };
+            img.src = src;
+        });
+    }
+
+    /* sniffImage decides which embedder to use from the bytes themselves
+       rather than from a mime string, which a fetched file may not carry */
+    function sniffImage(bytes) {
+        if (bytes.length > 3 && bytes[0] === 0xFF && bytes[1] === 0xD8) return "jpg";
+        return "png";
+    }
+
+    /* embedImage caches by source string: a deck that uses one picture on
+       twenty slides embeds its bytes once.
+
+       A native .ppta keeps its large media out of the body as media?file=
+       links, so a source that is not a data URL is fetched. Embedding the
+       original bytes is the point - re-encoding through a canvas would
+       turn a photo into a much larger lossless PNG. */
+    function makeImageEmbedder(pdfDoc) {
+        var cache = {};
+        function embedBytes(bytes) {
+            return sniffImage(bytes) === "jpg" ? pdfDoc.embedJpg(bytes) : pdfDoc.embedPng(bytes);
+        }
+        return function (src) {
+            if (!src) return Promise.resolve(null);
+            if (cache[src]) return cache[src];
+            var p;
+            var d = dataUrlBytes(src);
+            if (d) {
+                p = /jpe?g/i.test(d.mime) ? pdfDoc.embedJpg(d.bytes) : embedBytes(d.bytes);
+            } else {
+                p = fetch(src).then(function (r) {
+                    if (!r.ok) throw new Error("cannot read " + src);
+                    return r.arrayBuffer();
+                }).then(function (buf) {
+                    return embedBytes(new Uint8Array(buf));
+                });
+            }
+            cache[src] = p.catch(function () { return null; });
+            return cache[src];
+        };
+    }
+
+    return {
+        PX_TO_PT: PX_TO_PT,
+        RASTER_SCALE: RASTER_SCALE,
+        STD_FAMILIES: STD_FAMILIES,
+        winAnsiCp: winAnsiCp,
+        haveFamily: haveFamily,
+        familyList: familyList,
+        loadFontkit: loadFontkit,
+        makeFonts: makeFonts,
+        resolveChar: resolveChar,
+        segmentText: segmentText,
+        faceOf: faceOf,
+        fontMetricsOf: fontMetricsOf,
+        eachTextNode: eachTextNode,
+        px: px,
+        clamp: clamp,
+        parseFill: parseFill,
+        parseColor: parseColor,
+        contrastOf: contrastOf,
+        dataUrlBytes: dataUrlBytes,
+        Page: Page,
+        collectRuns: collectRuns,
+        splitByLine: splitByLine,
+        canDrawAsText: canDrawAsText,
+        drawFragment: drawFragment,
+        drawRuns: drawRuns,
+        rasterizeElement: rasterizeElement,
+        rasterizeFallback: rasterizeFallback,
+        inlineStyles: inlineStyles,
+        filteredImageData: filteredImageData,
+        sniffImage: sniffImage,
+        makeImageEmbedder: makeImageEmbedder
+    };
+})();

+ 286 - 0
src/web/Office/common/pdfdraw.js

@@ -0,0 +1,286 @@
+/*
+    ArozOS Office - PDF assembly from a display list
+    ================================================
+    The half of a browser-side PDF export that needs no page to look at.
+    The exporter measures the document in the page (where every line, box
+    and baseline is) and writes down what to draw as plain data - a display
+    list. This file turns that list into a PDF with pdf-lib.
+
+    It is written to run in a Web Worker (common/pdfworker.js), because this
+    is the part that takes the time: embedding and deflating every picture,
+    subsetting the fonts and serializing the file used to hold the page
+    still for most of a large export. Nothing here touches the DOM, so the
+    same code also runs in the page when a worker cannot be started.
+
+    The job:
+        {
+          title, sheetW, sheetH,              // px (96/in)
+          images: { id: { src } | { bytes } },  // data URL / URL, or JPEG/PNG bytes
+          pages: [ [op, ...], ... ]
+        }
+    Coordinates are px from the page's top-left corner. Ops are arrays:
+        ["rect", x, y, w, h, rgb, alpha]
+        ["image", id, x, y, w, h, crop]       crop {t,r,b,l} fractions | null
+        ["text", x, baseline, size, rgb, fitW, segs, deco]
+              segs [[text, font, synthBold]]  font "s:Helvetica:bold:italic" | "f:<url>"
+              deco [y, h, w] underline/strike bar, w used when nothing drew
+        ["leader", x, w, baseline, size, rgb, ch, font]
+        ["ellipse", cx, cy, rx, ry, rgb, strokeW]      strokeW 0 = filled
+        ["poly", [x, y, ...], rgb, strokeW]
+        ["frame", x, y, w, h, rgb, strokeW]
+
+    Usage:
+        OfficePdfDraw.render(job, { fontkit, onProgress }) -> Promise<Uint8Array>
+*/
+
+var OfficePdfDraw = (function () {
+    "use strict";
+
+    var PX = 0.75;
+    var STD_VARIANTS = {
+        Helvetica: ["Helvetica", "HelveticaBold", "HelveticaOblique", "HelveticaBoldOblique"],
+        TimesRoman: ["TimesRoman", "TimesRomanBold", "TimesRomanItalic", "TimesRomanBoldItalic"],
+        Courier: ["Courier", "CourierBold", "CourierOblique", "CourierBoldOblique"]
+    };
+
+    function rgb(c) {
+        return PDFLib.rgb(c[0], c[1], c[2]);
+    }
+
+    function base64Bytes(src) {
+        var comma = src.indexOf(",");
+        if (comma < 0 || src.substring(0, comma).indexOf(";base64") < 0) return null;
+        var bin = atob(src.substring(comma + 1));
+        var out = new Uint8Array(bin.length);
+        for (var i = 0; i < bin.length; i++) out[i] = bin.charCodeAt(i);
+        return out;
+    }
+    function sniff(bytes) {
+        if (bytes.length > 3 && bytes[0] === 0xFF && bytes[1] === 0xD8) return "jpg";
+        if (bytes.length > 8 && bytes[0] === 0x89 && bytes[1] === 0x50 && bytes[2] === 0x4E && bytes[3] === 0x47) return "png";
+        return null;
+    }
+
+    function render(job, opts) {
+        opts = opts || {};
+        var pdfDoc;
+        var fontCache = {};    // font ref -> Promise<PDFFont|null>
+        var fontReady = {};    // font ref -> PDFFont|null
+        var imageCache = {};   // id -> Promise<PDFImage|null>
+
+        function loadFont(ref) {
+            if (fontCache[ref]) return fontCache[ref];
+            var p;
+            if (ref.indexOf("s:") === 0) {
+                var parts = ref.split(":");
+                var names = STD_VARIANTS[parts[1]] || STD_VARIANTS.Helvetica;
+                var key = names[(parts[2] === "1" ? 1 : 0) + (parts[3] === "1" ? 2 : 0)];
+                p = Promise.resolve(pdfDoc.embedStandardFont(PDFLib.StandardFonts[key]));
+            } else {
+                var url = ref.substring(2);
+                p = fetch(url).then(function (r) {
+                    if (!r.ok) throw new Error("cannot read " + url);
+                    return r.arrayBuffer();
+                }).then(function (buf) {
+                    return pdfDoc.embedFont(new Uint8Array(buf), { subset: true });
+                });
+            }
+            fontCache[ref] = p.then(function (f) { fontReady[ref] = f; return f; }, function () {
+                fontReady[ref] = null;
+                return null;
+            });
+            return fontCache[ref];
+        }
+
+        function loadImage(id) {
+            if (imageCache[id]) return imageCache[id];
+            var spec = (job.images || {})[id] || {};
+            var bytesP;
+            if (spec.bytes) {
+                bytesP = Promise.resolve(spec.bytes);
+            } else if (spec.src && spec.src.indexOf("data:") === 0) {
+                bytesP = Promise.resolve(base64Bytes(spec.src));
+            } else if (spec.src) {
+                bytesP = fetch(spec.src).then(function (r) {
+                    if (!r.ok) throw new Error("cannot read " + spec.src);
+                    return r.arrayBuffer();
+                }).then(function (b) { return new Uint8Array(b); });
+            } else {
+                bytesP = Promise.resolve(null);
+            }
+            imageCache[id] = bytesP.then(function (bytes) {
+                if (!bytes) return null;
+                var kind = sniff(bytes);
+                if (kind === "jpg") return pdfDoc.embedJpg(bytes);
+                if (kind === "png") return pdfDoc.embedPng(bytes);
+                return null;
+            }).catch(function () { return null; });
+            return imageCache[id];
+        }
+
+        // everything a page refers to is loaded before it is drawn, so the
+        // drawing itself stays synchronous
+        function preload(ops) {
+            var jobs = [];
+            ops.forEach(function (op) {
+                if (op[0] === "image") jobs.push(loadImage(op[1]));
+                else if (op[0] === "text") op[6].forEach(function (s) { jobs.push(loadFont(s[1])); });
+                else if (op[0] === "leader") jobs.push(loadFont(op[7]));
+            });
+            return Promise.all(jobs);
+        }
+
+        function drawPage(page, ops, images) {
+            var H = job.sheetH;
+            var y = function (top) { return (H - top) * PX; };
+            var push = function (list) { page.pushOperators.apply(page, list); };
+            ops.forEach(function (op, idx) {
+                switch (op[0]) {
+                    case "rect":
+                        page.drawRectangle({
+                            x: op[1] * PX, y: y(op[2] + op[4]), width: op[3] * PX, height: op[4] * PX,
+                            color: rgb(op[5]), opacity: op[6]
+                        });
+                        break;
+                    case "frame":
+                        page.drawRectangle({
+                            x: op[1] * PX, y: y(op[2] + op[4]), width: op[3] * PX, height: op[4] * PX,
+                            borderColor: rgb(op[5]), borderWidth: op[6] * PX
+                        });
+                        break;
+                    case "image":
+                        var emb = images[idx];
+                        if (!emb) break;
+                        var b = { x: op[2], y: op[3], w: op[4], h: op[5] };
+                        var crop = op[6];
+                        push([PDFLib.pushGraphicsState()]);
+                        if (crop) {
+                            push([
+                                PDFLib.moveTo(b.x * PX, y(b.y)), PDFLib.lineTo((b.x + b.w) * PX, y(b.y)),
+                                PDFLib.lineTo((b.x + b.w) * PX, y(b.y + b.h)), PDFLib.lineTo(b.x * PX, y(b.y + b.h)),
+                                PDFLib.closePath(), PDFLib.clip(), PDFLib.endPath()
+                            ]);
+                            var kw = 1 - crop.l - crop.r, kh = 1 - crop.t - crop.b;
+                            if (kw > 0.001 && kh > 0.001) {
+                                var fw = b.w / kw, fh = b.h / kh;
+                                b = { x: b.x - crop.l * fw, y: b.y - crop.t * fh, w: fw, h: fh };
+                            }
+                        }
+                        page.drawImage(emb, { x: b.x * PX, y: y(b.y + b.h), width: b.w * PX, height: b.h * PX });
+                        push([PDFLib.popGraphicsState()]);
+                        break;
+                    case "text":
+                        drawText(page, op, y, push);
+                        break;
+                    case "leader":
+                        var font = fontReady[op[7]];
+                        if (!font) break;
+                        var cw = font.widthOfTextAtSize(op[6], op[4]);
+                        if (!(cw > 0)) break;
+                        var n = Math.floor((op[2] - cw) / cw);
+                        if (n < 1) break;
+                        page.drawText(new Array(n + 1).join(op[6]), {
+                            x: (op[1] + op[2] - n * cw) * PX, y: y(op[3]),
+                            size: op[4] * PX, font: font, color: rgb(op[5])
+                        });
+                        break;
+                    case "ellipse":
+                        page.drawEllipse(op[6] > 0 ? {
+                            x: op[1] * PX, y: y(op[2]), xScale: op[3] * PX, yScale: op[4] * PX,
+                            borderColor: rgb(op[5]), borderWidth: op[6] * PX
+                        } : {
+                            x: op[1] * PX, y: y(op[2]), xScale: op[3] * PX, yScale: op[4] * PX,
+                            color: rgb(op[5])
+                        });
+                        break;
+                    case "poly":
+                        var pts = op[1];
+                        if (pts.length < 4) break;
+                        var list = [PDFLib.pushGraphicsState()];
+                        list.push(op[3] > 0 ? PDFLib.setStrokingColor(rgb(op[2])) : PDFLib.setFillingColor(rgb(op[2])));
+                        if (op[3] > 0) list.push(PDFLib.setLineWidth(op[3] * PX));
+                        list.push(PDFLib.moveTo(pts[0] * PX, y(pts[1])));
+                        for (var i = 2; i + 1 < pts.length; i += 2) list.push(PDFLib.lineTo(pts[i] * PX, y(pts[i + 1])));
+                        list.push(PDFLib.closePath(), op[3] > 0 ? PDFLib.stroke() : PDFLib.fill(), PDFLib.popGraphicsState());
+                        push(list);
+                        break;
+                }
+            });
+        }
+
+        /* one line fragment, in as many pieces as it takes fonts to spell
+           it, squeezed or stretched to the width the browser gave it */
+        function drawText(page, op, y, push) {
+            var x = op[1], baseline = op[2], size = op[3], col = rgb(op[4]), fitW = op[5];
+            var segs = [];
+            var total = 0;
+            for (var i = 0; i < op[6].length; i++) {
+                var s = op[6][i];
+                var font = fontReady[s[1]];
+                if (!font) return;
+                var w = font.widthOfTextAtSize(s[0], size);
+                segs.push({ text: s[0], font: font, synthBold: s[2], w: w });
+                total += w;
+            }
+            var scale = 1;
+            if (fitW > 0 && total > 0) {
+                var ratio = fitW / total;
+                // a ratio far from 1 means the measurement, not the font, is
+                // wrong (a collapsed space, a transform) - leave it alone
+                if (ratio > 0.5 && ratio < 2 && Math.abs(ratio - 1) > 0.005) scale = ratio;
+            }
+            var cursor = x;
+            segs.forEach(function (seg) {
+                var ops = [PDFLib.pushGraphicsState()];
+                if (scale !== 1) ops.push(PDFLib.setCharacterSqueeze(scale * 100));
+                if (seg.synthBold) {
+                    ops.push(PDFLib.setTextRenderingMode(PDFLib.TextRenderingMode.FillAndOutline));
+                    ops.push(PDFLib.setLineWidth(size / 28 * PX));
+                    ops.push(PDFLib.setStrokingColor(col));
+                }
+                push(ops);
+                page.drawText(seg.text, { x: cursor * PX, y: y(baseline), size: size * PX, font: seg.font, color: col });
+                push([PDFLib.popGraphicsState()]);
+                cursor += seg.w * scale;
+            });
+            var deco = op[7];
+            if (deco) {
+                var dw = total * scale || deco[2];
+                page.drawRectangle({ x: x * PX, y: y(deco[0] + deco[1]), width: dw * PX, height: deco[1] * PX, color: col });
+            }
+        }
+
+        function tick() {
+            return new Promise(function (res) { setTimeout(res, 0); });
+        }
+
+        var total = job.pages.length;
+        return PDFLib.PDFDocument.create().then(function (doc) {
+            pdfDoc = doc;
+            if (opts.fontkit) doc.registerFontkit(opts.fontkit);
+            if (job.title) doc.setTitle(job.title);
+            doc.setProducer("ArozOS Office");
+            var chain = Promise.resolve();
+            job.pages.forEach(function (ops, i) {
+                chain = chain.then(function () {
+                    return preload(ops);
+                }).then(function () {
+                    return Promise.all(ops.map(function (op) {
+                        return op[0] === "image" ? loadImage(op[1]) : null;
+                    }));
+                }).then(function (images) {
+                    var page = pdfDoc.addPage([job.sheetW * PX, job.sheetH * PX]);
+                    drawPage(page, ops, images);
+                    if (opts.onProgress) opts.onProgress(i + 1, total, "page");
+                    return tick();
+                });
+            });
+            return chain;
+        }).then(function () {
+            if (opts.onProgress) opts.onProgress(total, total, "save");
+            return pdfDoc.save({ objectsPerTick: 20 });
+        });
+    }
+
+    return { render: render };
+})();

+ 28 - 0
src/web/Office/common/pdfworker.js

@@ -0,0 +1,28 @@
+/*
+    ArozOS Office - PDF worker
+    ==========================
+    Assembles a PDF from a display list off the page's thread (see
+    pdfdraw.js), so a large export does not freeze the editor.
+
+    Messages in:   { job }
+    Messages out:  { type: "progress", done, total, stage }
+                   { type: "done", bytes }       (bytes transferred)
+                   { type: "error", message }
+*/
+/* global importScripts, OfficePdfDraw, fontkit */
+importScripts("lib/pdf-lib.min.js", "lib/fontkit.umd.min.js", "pdfdraw.js");
+
+self.onmessage = function (e) {
+    var job = e.data && e.data.job;
+    if (!job) return;
+    OfficePdfDraw.render(job, {
+        fontkit: self.fontkit,
+        onProgress: function (done, total, stage) {
+            self.postMessage({ type: "progress", done: done, total: total, stage: stage });
+        }
+    }).then(function (bytes) {
+        self.postMessage({ type: "done", bytes: bytes }, [bytes.buffer]);
+    }).catch(function (err) {
+        self.postMessage({ type: "error", message: (err && err.message) || String(err) });
+    });
+};

+ 215 - 42
src/web/Office/docs/docs.css

@@ -23,16 +23,37 @@
     display: flex;
     display: flex;
     flex-direction: column;
     flex-direction: column;
     box-sizing: border-box;
     box-sizing: border-box;
-    background: #ffffff;
+    /* #page is the frame the flow is laid out in; the paper is drawn by one
+       .doc-sheet per page behind it, so the gap between two sheets is real
+       empty space rather than a band painted over one long sheet */
+    background: transparent;
     color: #1f2328;
     color: #1f2328;
-    /*box-shadow: 0 1px 3px var(--of-paper-shadow), 0 4px 14px var(--of-paper-shadow); */
-    border: 1px solid #e2e5e9;
-    border-radius: 2px;
 }
 }
-body.dark #page {
+#editor {
+    position: relative;
+    z-index: 1;
+}
+#pageSheets {
+    position: absolute;
+    left: 0;
+    right: 0;
+    top: 0;
+    height: 0;
+    z-index: 0;
+    pointer-events: none;
+}
+.doc-sheet {
+    position: absolute;
+    left: 0;
+    right: 0;
     /* paper stays white in dark mode - deliberate, see header comment */
     /* paper stays white in dark mode - deliberate, see header comment */
     background: #ffffff;
     background: #ffffff;
-    color: #1f2328;
+    /* an outline, not a border: the sheet is exactly the page's size */
+    outline: 1px solid #e2e5e9;
+    border-radius: 2px;
+}
+body.dark .doc-sheet {
+    outline: none;
     box-shadow: 0 0 0 1px rgba(255, 255, 255, 0.06), 0 6px 22px rgba(0, 0, 0, 0.6);
     box-shadow: 0 0 0 1px rgba(255, 255, 255, 0.06), 0 6px 22px rgba(0, 0, 0, 0.6);
 }
 }
 
 
@@ -51,17 +72,15 @@ body.dark #page {
     font-family: Arial, Helvetica, sans-serif;
     font-family: Arial, Helvetica, sans-serif;
     font-size: 9pt;
     font-size: 9pt;
     color: #6b7078;
     color: #6b7078;
-    padding: 2px 0;
-    border-bottom: 1px dashed transparent;
+    padding: 0;
     cursor: text;
     cursor: text;
     z-index: 6;
     z-index: 6;
 }
 }
-.doc-hf.doc-hf-footer {
-    border-bottom: none;
-    border-top: 1px dashed transparent;
-}
+/* the edit affordance is an outline: a border would move the content off
+   the header/footer distance the document states */
 .doc-hf:hover, .doc-hf:focus {
 .doc-hf:hover, .doc-hf:focus {
-    border-color: #c9cdd3;
+    outline: 1px dashed #c9cdd3;
+    outline-offset: 1px;
 }
 }
 /* only the first visible copy nags: one placeholder per document, not one
 /* only the first visible copy nags: one placeholder per document, not one
    per page */
    per page */
@@ -71,22 +90,67 @@ body.dark #page {
     pointer-events: none;
     pointer-events: none;
 }
 }
 .doc-hf[hidden] { display: none; }
 .doc-hf[hidden] { display: none; }
+/* rich header/footer content (an imported header, a picture) is document
+   text: it takes the body's typography, not the placeholder grey */
+.doc-hf > p, .doc-hf > h1, .doc-hf > h2, .doc-hf > h3, .doc-hf > h4,
+.doc-hf > h5, .doc-hf > h6, .doc-hf > table, .doc-hf > ul, .doc-hf > ol {
+    margin: 0;
+    font-family: Arial, "Noto Sans", "Noto Sans TC", "Noto Sans SC", "Noto Sans JP", "Noto Sans KR", sans-serif;
+    font-size: 11pt;
+    color: #000000;
+}
+.doc-hf img { max-width: none; }
+.doc-hf img.doc-anchor { display: block; }
+.doc-hf .doc-field { font-variant-numeric: tabular-nums; }
+
+/* the page number of "Page setup > Print page numbers" (no PAGE field) */
+.doc-pagenum {
+    position: absolute;
+    text-align: center;
+    font-family: Arial, "Noto Sans", sans-serif;
+    font-size: 10pt;
+    color: #444444;
+    pointer-events: none;
+    z-index: 6;
+}
 
 
 /* ============ Editor content ============ */
 /* ============ Editor content ============ */
+/* Typography follows the word processors a .docx comes from: spacing
+   before/after is margin (neighbouring spacing collapses to the larger of
+   the two, as it does in Google Docs), and every block's line height is set in px by docs_layout.js from the
+   font's own metrics. Inline elements get line-height 0 so a run in
+   another font cannot make its line taller than that rule says - the
+   layout engine gives a bigger run its own height. */
 #editor {
 #editor {
     flex: 1 1 auto;
     flex: 1 1 auto;
     outline: none;
     outline: none;
-    font-family: Arial, Helvetica, sans-serif;
+    font-family: Arial, "Noto Sans", "Noto Sans TC", "Noto Sans SC", "Noto Sans JP", "Noto Sans KR", sans-serif;
     font-size: 11pt;
     font-size: 11pt;
-    line-height: 1.5;
+    line-height: 1.32;
+    color: #000000;
+    tab-size: 36pt;
     word-wrap: break-word;
     word-wrap: break-word;
     overflow-wrap: break-word;
     overflow-wrap: break-word;
     caret-color: #1a73e8;
     caret-color: #1a73e8;
 }
 }
-#editor p, #editor div { margin: 0 0 0pt; }
-#editor h1, #editor h2, #editor h3, #editor h4 {
+#editor span, #editor a, #editor b, #editor strong, #editor i, #editor em,
+#editor u, #editor s, #editor strike, #editor font, #editor code,
+#editor sup, #editor sub, #editor mark, #editor small,
+.doc-hf span, .doc-hf a, .doc-hf b, .doc-hf i, .doc-hf u, .doc-hf sup, .doc-hf sub,
+.doc-fn span, .doc-fn a, .doc-fn b, .doc-fn i, .doc-fn u, .doc-fn sup, .doc-fn sub {
+    line-height: 0;
+}
+#editor sup, #editor sub, .doc-hf sup, .doc-hf sub, .doc-fn sup, .doc-fn sub {
+    font-size: 60%;
+    vertical-align: 0.667em;
+}
+#editor sub, .doc-hf sub, .doc-fn sub { vertical-align: -0.233em; }
+#editor p, #editor div { margin: 0; }
+#editor h1, #editor h2, #editor h3, #editor h4, #editor h5, #editor h6 {
     margin: 14pt 0 6pt;
     margin: 14pt 0 6pt;
-    line-height: 1.25;
+    /* the page's typeface, not the UI kit's heading font (the export
+       writes headings in the body font) */
+    font-family: inherit;
     font-weight: 600;
     font-weight: 600;
     color: #1f2328;
     color: #1f2328;
 }
 }
@@ -94,11 +158,18 @@ body.dark #page {
 #editor h2 { font-size: 16pt; }
 #editor h2 { font-size: 16pt; }
 #editor h3 { font-size: 13pt; }
 #editor h3 { font-size: 13pt; }
 #editor h4 { font-size: 11pt; font-style: italic; }
 #editor h4 { font-size: 11pt; font-style: italic; }
+#editor h5 { font-size: 11pt; }
+#editor h6 { font-size: 11pt; font-style: italic; }
 #editor .doc-title {
 #editor .doc-title {
     font-size: 26pt;
     font-size: 26pt;
     font-weight: 400;
     font-weight: 400;
     margin: 0 0 12pt;
     margin: 0 0 12pt;
 }
 }
+#editor .doc-subtitle {
+    font-size: 15pt;
+    color: #666666;
+    margin: 0 0 16pt;
+}
 #editor a { color: #1a58c2; text-decoration: underline; cursor: text; }
 #editor a { color: #1a58c2; text-decoration: underline; cursor: text; }
 #editor a:hover { color: #0f3f96; }
 #editor a:hover { color: #0f3f96; }
 #editor hr {
 #editor hr {
@@ -124,8 +195,73 @@ body.dark #page {
     white-space: pre-wrap;
     white-space: pre-wrap;
     margin: 8pt 0;
     margin: 8pt 0;
 }
 }
-#editor ul, #editor ol { margin: 4pt 0 8pt; padding-left: 28px; }
-#editor li { margin: 2pt 0; }
+/* Lists: the marker text is computed by docs_layout.js (li[data-marker])
+   and hangs --doc-hang to the left of the item's text, which starts at the
+   list's padding - Word's "indent" and "hanging" */
+#editor ul, #editor ol, .doc-hf ul, .doc-hf ol, .doc-fn ul, .doc-fn ol {
+    margin: 0;
+    padding-left: 36pt;
+    list-style: none;
+}
+#editor li, .doc-hf li, .doc-fn li { position: relative; margin: 0; }
+#editor li::before, .doc-hf li::before, .doc-fn li::before {
+    content: attr(data-marker);
+    position: absolute;
+    left: calc(-1 * var(--doc-hang, 18pt));
+    white-space: pre;
+    text-indent: 0;
+    font-style: normal;
+    text-decoration: none;
+}
+
+/* Tabs: a real tab character, sized to its tab stop by docs_layout.js */
+span.doc-tab { white-space: pre; }
+span.doc-tab[data-leader="dot"] {
+    background-image: radial-gradient(circle, currentColor 0.55px, transparent 0.8px);
+    background-size: 4px 3px;
+    background-repeat: repeat-x;
+    background-position: 0 calc(100% - 0.28em);
+}
+span.doc-tab[data-leader="hyphen"], span.doc-tab[data-leader="underscore"] {
+    border-bottom: 1px solid currentColor;
+}
+
+/* Footnote references in the text */
+#editor sup.doc-fnref { cursor: default; user-select: none; }
+
+/* Layout-only spacers the paginator inserts (never saved) */
+#editor .doc-autobreak {
+    display: block;
+    margin: 0;
+    padding: 0;
+    border: 0;
+    user-select: none;
+    -webkit-user-select: none;
+    pointer-events: none;
+}
+#editor tr.doc-autobreak { display: table-row; }
+#editor td.doc-autobreak-cell {
+    padding: 0 !important;
+    border: none !important;
+    background: transparent !important;
+    line-height: 0;
+}
+/* The two halves of a block split across a page (docs_layout.js): the
+   spacing, rule and first-line indent belong to the paragraph's start and
+   end, not to the page cut, and the continuation of a list item has no
+   marker of its own. */
+#editor .doc-split-head {
+    margin-bottom: 0 !important;
+    padding-bottom: 0 !important;
+    border-bottom-style: none !important;
+}
+#editor .doc-split-tail {
+    margin-top: 0 !important;
+    padding-top: 0 !important;
+    border-top-style: none !important;
+    text-indent: 0 !important;
+}
+#editor li.doc-split-tail::before { content: none !important; }
 
 
 /* Images */
 /* Images */
 #editor img {
 #editor img {
@@ -133,6 +269,12 @@ body.dark #page {
     height: auto;
     height: auto;
     cursor: default;
     cursor: default;
 }
 }
+/* a picture anchored above/below the text (a full-bleed cover, a banner):
+   its own box on its own line, allowed past the text column */
+#editor img.doc-anchor {
+    display: block;
+    max-width: none;
+}
 #editor img.of-selimg {
 #editor img.of-selimg {
     outline: 2px solid var(--of-accent);
     outline: 2px solid var(--of-accent);
     outline-offset: 1px;
     outline-offset: 1px;
@@ -155,14 +297,52 @@ body.dark #page {
     border-collapse: collapse;
     border-collapse: collapse;
     width: 100%;
     width: 100%;
     margin: 8pt 0;
     margin: 8pt 0;
+    /* columns keep their widths (equal until resized) whatever is typed
+       into them - the way Word and the export lay a table out */
+    table-layout: fixed;
 }
 }
 #editor table.of-table td, #editor table.of-table th {
 #editor table.of-table td, #editor table.of-table th {
     border: 1px solid #b9bec7;
     border: 1px solid #b9bec7;
     padding: 4px 8px;
     padding: 4px 8px;
     min-width: 2em;
     min-width: 2em;
     vertical-align: top;
     vertical-align: top;
+    overflow-wrap: break-word;
 }
 }
 #editor table.of-table th { background: #f1f3f4; font-weight: 600; }
 #editor table.of-table th { background: #f1f3f4; font-weight: 600; }
+/* an imported table states its own borders, padding and width inline */
+#editor table.of-table[data-docx] { margin-top: 0; margin-bottom: 0; }
+#editor table.of-table[data-docx] td, #editor table.of-table[data-docx] th { min-width: 0; }
+
+/* Footnotes: drawn at the bottom of the page that references them */
+.doc-fn-area {
+    position: absolute;
+    z-index: 6;
+    font-family: Arial, "Noto Sans", "Noto Sans TC", "Noto Sans SC", "Noto Sans JP", "Noto Sans KR", sans-serif;
+    font-size: 10pt;
+    color: #000000;
+}
+.doc-fn-sep {
+    height: 10.8pt;
+    position: relative;
+}
+.doc-fn-sep::after {
+    content: "";
+    position: absolute;
+    left: 0;
+    top: 5.02pt;
+    width: 144pt;
+    border-top: 0.75pt solid #000000;
+}
+.doc-fn { outline: none; }
+.doc-fn p { margin: 0; }
+.doc-fn sup.doc-fnnum { user-select: none; }
+.doc-fn-measure {
+    position: absolute;
+    left: 0;
+    top: 0;
+    visibility: hidden;
+    pointer-events: none;
+}
 
 
 /* Multi-column layout: title/author blocks marked via
 /* Multi-column layout: title/author blocks marked via
    Format > Page layout > Span all columns stretch across every column */
    Format > Page layout > Span all columns stretch across every column */
@@ -197,12 +377,10 @@ body.dark #page {
 }
 }
 
 
 /* ============ Page breaks (explicit + automatic pagination) ============
 /* ============ Page breaks (explicit + automatic pagination) ============
-   Real blocks in the flow, stretched by updatePageGuides() to eat the rest
-   of their sheet. The .doc-pb-gap child paints the cut between the two
-   sheets: it reaches PAST the sheet edges (GAP_OVERHANG_PX) so it also
-   erases the sheet's side shadow across the gap - each page keeps its own
-   separate vertical shadow. The .doc-pb-gap-in child draws the two facing
-   sheet edges across the actual sheet width only. */
+   An explicit break is a real block in the flow, stretched by the
+   paginator to eat the rest of its sheet; so is every automatic spacer.
+   Nothing is painted over the gap between two sheets - there is simply no
+   sheet there (#pageSheets). */
 .doc-pagebreak {
 .doc-pagebreak {
     position: relative;
     position: relative;
     margin: 0;
     margin: 0;
@@ -212,25 +390,10 @@ body.dark #page {
     break-after: page;
     break-after: page;
     page-break-after: always;
     page-break-after: always;
 }
 }
-.doc-pb-gap {
-    position: absolute;
-    background: var(--of-bg);
-    pointer-events: none;
-}
-.doc-pb-gap-in {
-    position: absolute;
-    top: 0;
-    bottom: 0;
-    /* left/right set inline to the overhang so edges align with the sheet */
-    /* box-shadow: inset 0 4px 6px -4px var(--of-paper-shadow),
-                inset 0 -4px 6px -4px var(--of-paper-shadow); */
-    border-top: 1px solid rgba(60, 64, 67, 0.16);
-    border-bottom: 1px solid rgba(60, 64, 67, 0.16);
-}
 @media print {
 @media print {
     /* the sheet cut is a screen affordance; the printer paginates for real */
     /* the sheet cut is a screen affordance; the printer paginates for real */
     .doc-pagebreak { height: 0 !important; }
     .doc-pagebreak { height: 0 !important; }
-    .doc-pb-gap { display: none; }
+    #pageSheets { display: none; }
     /* automatic spacers are layout-only - the printer re-paginates */
     /* automatic spacers are layout-only - the printer re-paginates */
     .doc-autobreak { display: none !important; }
     .doc-autobreak { display: none !important; }
 }
 }
@@ -562,3 +725,13 @@ body.dark #editor del.doc-del { color: #f28b82; background: rgba(242, 139, 130,
 @media print {
 @media print {
     #docCmtPanel { display: none !important; }
     #docCmtPanel { display: none !important; }
 }
 }
+
+/* ============ PDF export state ============
+   While docs_pdf.js draws the pages, review marks show as resolved: the
+   suggestions accepted and the comment highlights off - what every other
+   export writes too. (Hiding deletions changes the flow, so the export
+   repaginates after setting this.) */
+body.doc-exporting #editor span.doc-cmt { background: none; border-bottom: 0; }
+body.doc-exporting #editor ins.doc-ins { color: inherit; text-decoration: inherit; background: none; }
+body.doc-exporting #editor del.doc-del { display: none; }
+body.doc-exporting .doc-hf-lead:empty::before { content: none; }

File diff suppressed because it is too large
+ 546 - 239
src/web/Office/docs/docs.js


+ 1662 - 0
src/web/Office/docs/docs_layout.js

@@ -0,0 +1,1662 @@
+/*
+    ArozOS Office - Docs layout engine
+    ==================================
+    Everything about how a document is laid out that CSS cannot decide on
+    its own, so that the page the editor shows is the page a word processor
+    would print - and, because the PDF exporter draws this very DOM, the
+    page the PDF gets.
+
+    Four passes over a rendered subtree (the editor, a header/footer copy,
+    a footnote area):
+
+      lineHeights  Word and Google Docs size a line from the font's own
+                   ascent and descent, rounded to whole pixels one side at
+                   a time: ceil(ascent x size x spacing) + ceil(descent x
+                   size x spacing). CSS line-height is size x number and
+                   drifts by a pixel or two a line, which over a page is a
+                   different page break. Every block that holds text gets
+                   that exact line height in px, and inline elements get
+                   line-height 0 (docs.css) so a run in another font cannot
+                   make its line taller than the rule says; a run in a
+                   bigger size gets its own px height and grows its line.
+      numbering    list markers ("1.", "b)", bullets) are computed here into
+                   li[data-marker] and drawn by li::before - a real value
+                   the PDF exporter can read, formats CSS counters do not
+                   have ("%1.%2."), and numbering that carries on across a
+                   list interrupted by a paragraph (same data-num).
+      tabs         span.doc-tab is sized to reach the next tab stop: the
+                   paragraph's own stops (data-tabs, right/center/left, with
+                   leaders) or the default 36pt grid.
+      footnotes    references are numbered in document order.
+
+    And pagination (paginate): the document is one contenteditable flow,
+    and page boundaries are made real in it. What crosses a boundary is
+    split into two elements - a paragraph at a line, the list or quote
+    around it, a table row cell by cell - and a spacer element between the
+    halves pushes the second one to the top of the next sheet, so long
+    paragraphs, lists and tables break where a word processor breaks them
+    and every half is a box of its own (borders and shading end at the
+    page). Spacers carry .doc-autobreak; splits are undone, and spacers
+    removed, on everything that is saved (docs.js cleanedHtml). Headings keep with the paragraph after
+    them, data-keep-lines paragraphs do not split, data-widow paragraphs
+    keep two lines on each side, footnotes take their space at the bottom of
+    the page that references them.
+
+    Coordinates are layout px relative to #page's padding box (offsetTop
+    space, unaffected by the framework's CSS zoom).
+*/
+
+var DocsLayout = (function () {
+    "use strict";
+
+    var PT = 96 / 72;           // css px per point
+    var MM = 96 / 25.4;         // css px per millimetre
+    var DEFAULT_TAB_PT = 36;
+
+    /* ---------------- font metrics ---------------- */
+
+    /* ascent/descent per em of the font a family list actually resolves to.
+       A canvas at 2048px reports the font's own units exactly (ArialMT:
+       1854/434), and it resolves a font stack the same way the text did. */
+    var ratioCache = {};
+    var ratioCtx = null;
+    /* Fonts a browser reports taller metrics for than the word processor
+       lays their lines out with. Consolas is the one that matters: in
+       Google Docs a Consolas code line is exactly as tall as an Arial line,
+       while its OS/2 win metrics would overshoot that by a pixel a line. */
+    var LINE_METRICS = {
+        "consolas": { asc: 1854 / 2048, desc: 434 / 2048 }
+    };
+    function fontRatios(family) {
+        var key = family || "";
+        if (ratioCache[key]) return ratioCache[key];
+        var out = { asc: 0.905, desc: 0.212 };
+        try {
+            if (!ratioCtx) ratioCtx = document.createElement("canvas").getContext("2d");
+            ratioCtx.font = "2048px " + (family || "sans-serif");
+            var m = ratioCtx.measureText("Hxg");
+            if (m && m.fontBoundingBoxAscent > 0) {
+                out = { asc: m.fontBoundingBoxAscent / 2048, desc: m.fontBoundingBoxDescent / 2048 };
+            }
+            var first = String(family || "").split(",")[0].trim().replace(/^["']|["']$/g, "").toLowerCase();
+            if (LINE_METRICS[first] && Math.abs(m.fontBoundingBoxAscent - 1884) < 2) out = LINE_METRICS[first];
+        } catch (e) { /* estimate */ }
+        ratioCache[key] = out;
+        return out;
+    }
+
+    // the word-processor line height rule, in px
+    function lineHeightPx(sizePx, family, mult) {
+        var r = fontRatios(family);
+        var s = sizePx * (mult > 0 ? mult : 1);
+        return Math.ceil(r.asc * s - 1e-6) + Math.ceil(r.desc * s - 1e-6);
+    }
+
+    /* ---------------- helpers ---------------- */
+
+    var INLINE_TAGS = {
+        A: 1, ABBR: 1, B: 1, BDI: 1, BDO: 1, BR: 1, CITE: 1, CODE: 1, DATA: 1, DEL: 1,
+        DFN: 1, EM: 1, FONT: 1, I: 1, IMG: 1, INS: 1, KBD: 1, MARK: 1, Q: 1, S: 1,
+        SAMP: 1, SMALL: 1, SPAN: 1, STRIKE: 1, STRONG: 1, SUB: 1, SUP: 1, TIME: 1,
+        TT: 1, U: 1, VAR: 1, WBR: 1
+    };
+    function isInlineEl(el) {
+        if (!INLINE_TAGS[el.tagName]) return false;
+        // an anchored picture or a line spacer is display:block
+        if (el.classList.contains("doc-anchor") || el.classList.contains("doc-autobreak")) return false;
+        return true;
+    }
+    function isSpacer(el) {
+        return el.nodeType === 1 && el.classList.contains("doc-autobreak");
+    }
+
+    // does this element hold inline content of its own (a "line block")?
+    function holdsInline(el) {
+        for (var c = el.firstChild; c; c = c.nextSibling) {
+            if (c.nodeType === 3) {
+                if (c.nodeValue && c.nodeValue.length) return true;
+            } else if (c.nodeType === 1 && isInlineEl(c)) {
+                return true;
+            }
+        }
+        return false;
+    }
+    function hasBlockChild(el) {
+        for (var c = el.firstElementChild; c; c = c.nextElementSibling) {
+            if (!isInlineEl(c) && !isSpacer(c)) return true;
+        }
+        return false;
+    }
+
+    // multiple of single spacing stated on a block, or the legacy
+    // unitless line-height an older document carries
+    function blockSpacing(el, defLS) {
+        var ls = parseFloat(el.getAttribute("data-ls"));
+        if (ls > 0) return ls;
+        // documents from before data-ls said it as a unitless line-height
+        var inline = el.style.lineHeight;
+        if (inline && /^[\d.]+$/.test(inline) && parseFloat(inline) > 0) {
+            el.setAttribute("data-ls", inline);
+            return parseFloat(inline);
+        }
+        return defLS;
+    }
+
+    function cssLenPx(v) {
+        v = String(v || "").trim();
+        var n = parseFloat(v);
+        if (isNaN(n)) return 0;
+        if (/pt$/.test(v)) return n * PT;
+        if (/mm$/.test(v)) return n * MM;
+        if (/in$/.test(v)) return n * 96;
+        return n;
+    }
+
+    /* ---------------- line heights ---------------- */
+
+    /* lineHeights sets the px line height of every block in root that
+       holds text, and the own height of any run that is bigger than its
+       block's smallest run. */
+    function applyLineHeights(root, defLS) {
+        if (!root) return;
+        defLS = defLS > 0 ? defLS : 1.15;
+        var blocks = [];
+        if (holdsInline(root) || !root.firstElementChild) blocks.push(root);
+        var all = root.getElementsByTagName("*");
+        for (var i = 0; i < all.length; i++) {
+            var el = all[i];
+            if (isInlineEl(el) || isSpacer(el)) continue;
+            if (el.tagName === "TABLE" || el.tagName === "TBODY" || el.tagName === "TR" ||
+                el.tagName === "COLGROUP" || el.tagName === "COL") continue;
+            if (holdsInline(el) || !el.firstElementChild) blocks.push(el);
+        }
+        blocks.forEach(function (b) { lineHeightFor(b, defLS); });
+    }
+
+    function lineHeightFor(block, defLS) {
+        var bcs = window.getComputedStyle(block);
+        if (bcs.display === "none") return;
+        var exact = block.getAttribute("data-lsexact");
+        if (exact) {
+            block.style.lineHeight = cssLenPx(exact) + "px";
+            return;
+        }
+        var mult = blockSpacing(block, defLS);
+        var minPx = cssLenPx(block.getAttribute("data-lsmin"));
+
+        // the runs directly in this block: their sizes decide the lines
+        var runs = [];
+        var walker = document.createTreeWalker(block, NodeFilter.SHOW_TEXT, {
+            acceptNode: function (n) {
+                // text of a nested block belongs to that block
+                for (var p = n.parentNode; p && p !== block; p = p.parentNode) {
+                    if (p.nodeType === 1 && !isInlineEl(p)) return NodeFilter.FILTER_REJECT;
+                }
+                return NodeFilter.FILTER_ACCEPT;
+            }
+        });
+        var preWs = bcs.whiteSpace.indexOf("pre") === 0 || bcs.whiteSpace === "break-spaces";
+        var node;
+        while ((node = walker.nextNode())) {
+            var txt = node.nodeValue;
+            if (!txt || (!preWs && !/\S/.test(txt))) continue;
+            var host = node.parentNode;
+            // superscript / subscript never make a line taller
+            var inScript = false;
+            for (var q = host; q && q !== block; q = q.parentNode) {
+                if (q.tagName === "SUP" || q.tagName === "SUB") { inScript = true; break; }
+            }
+            if (inScript) continue;
+            var cs = window.getComputedStyle(host);
+            runs.push({ el: host, size: parseFloat(cs.fontSize) || 14.67, family: cs.fontFamily });
+        }
+        var baseSize = parseFloat(bcs.fontSize) || 14.67;
+        var baseFamily = bcs.fontFamily;
+        var strut;
+        if (!runs.length) {
+            strut = lineHeightPx(baseSize, baseFamily, mult);
+        } else {
+            var min = runs[0];
+            runs.forEach(function (r) { if (r.size < min.size) min = r; });
+            // the paragraph's own font (its mark) takes part in every line
+            strut = Math.max(lineHeightPx(min.size, min.family, mult),
+                baseSize <= min.size + 0.01 ? lineHeightPx(baseSize, baseFamily, mult) : 0);
+            // bigger runs carry their own height so only their lines grow
+            runs.forEach(function (r) {
+                if (r.el === block) return;
+                if (r.size > min.size + 0.01) {
+                    var own = Math.max(lineHeightPx(r.size, r.family, mult), minPx);
+                    if (r.el.style.lineHeight !== own + "px") r.el.style.lineHeight = own + "px";
+                } else if (r.el.style.lineHeight) {
+                    r.el.style.lineHeight = "";
+                }
+            });
+        }
+        if (minPx > strut) strut = minPx;
+        var v = strut + "px";
+        if (block.style.lineHeight !== v) block.style.lineHeight = v;
+        if (!runs.length) pictureLines(block, strut, baseSize, baseFamily, mult);
+    }
+
+    /* A line holding pictures is as tall as the tallest picture, plus 1.75pt
+       above it and the part of a text line that hangs below the baseline
+       under it - the pictures sit on the baseline. In a paragraph that holds
+       only pictures that is set exactly: each picture is aligned to the
+       bottom of its line and carries those two amounts as margins, so the
+       line comes out the word processor's height to the fraction of a pixel
+       (baseline alignment would round the picture's top to a whole pixel,
+       which over a page of screenshots is a paragraph's worth of drift).
+       Where the text baseline sits is the one thing CSS and a word
+       processor disagree on: CSS splits the leading evenly around the
+       glyphs, Word and Google Docs put nearly all the leading that 1.15
+       spacing adds below them. The marks are undone before saving
+       (data-picline). */
+    var PICTURE_TOP_PX = 1.75 * PT;
+    function pictureLines(block, strut, sizePx, family, mult) {
+        var imgs = block.getElementsByTagName("img");
+        if (!imgs.length) return;
+        var r = fontRatios(family);
+        var single = lineHeightPx(sizePx, family, 1);
+        var baseline = Math.ceil(r.asc * sizePx - 1e-6) + 0.2 * (strut - single);
+        var below = Math.max(0, strut - baseline);
+        for (var i = 0; i < imgs.length; i++) {
+            var im = imgs[i];
+            if (im.classList.contains("doc-anchor")) continue;
+            if (im.parentNode !== block && !isInlineEl(im.parentNode)) continue;
+            im.setAttribute("data-picline", "1");
+            im.style.verticalAlign = "bottom";
+            im.style.marginTop = PICTURE_TOP_PX + "px";
+            im.style.marginBottom = below + "px";
+        }
+    }
+    function restorePictureLines(root) {
+        var imgs = root.querySelectorAll("img[data-picline]");
+        for (var i = 0; i < imgs.length; i++) {
+            imgs[i].style.verticalAlign = "";
+            imgs[i].style.marginTop = "";
+            imgs[i].style.marginBottom = "";
+            imgs[i].removeAttribute("data-picline");
+            if (!imgs[i].getAttribute("style")) imgs[i].removeAttribute("style");
+        }
+    }
+
+    /* ---------------- table borders ----------------
+       A browser draws a border in whole device pixels, so a 1pt (1.33px)
+       table rule takes 1px of layout - and a table of 40 rows comes out 13px
+       shorter than on paper. The difference goes back in as cell padding,
+       which is invisible and keeps the rows exactly as tall as the document
+       says. The padding the document states is kept in data-pad0 and put
+       back before anything is saved. */
+    function applyTableBorders(root) {
+        if (!root) return;
+        var tables = root.getElementsByTagName("table");
+        for (var t = 0; t < tables.length; t++) {
+            var tbl = tables[t];
+            // a table whose size has not moved since it was last fixed up
+            // needs nothing (the signature is a property, never saved)
+            if (tbl.__docsSig && tbl.__docsSig === tableSignature(tbl)) continue;
+            fixTable(tbl);
+            if (tbl.hasAttribute("data-docx")) roundRows(tbl);
+            tbl.__docsSig = tableSignature(tbl);
+        }
+    }
+    function tableSignature(tbl) {
+        return tbl.rows.length + ":" + tbl.offsetHeight + ":" + tbl.offsetWidth + ":" + (scaleOf(tbl) || 1).toFixed(3);
+    }
+    function fixTable(tbl) {
+        var rows = tbl.rows;
+        var cells = [];
+        // put back the stated padding first (writes only) ...
+        for (var r = 0; r < rows.length; r++) {
+            if (rows[r].classList.contains("doc-autobreak") || rows[r].closest("table") !== tbl) continue;
+            for (var c = 0; c < rows[r].cells.length; c++) {
+                var td = rows[r].cells[c];
+                var orig = td.getAttribute("data-pad0");
+                if (orig === null) {
+                    orig = td.style.padding || "";
+                    td.setAttribute("data-pad0", orig);
+                } else if (td.style.padding !== orig) {
+                    td.style.padding = orig;
+                }
+                cells.push({ td: td, last: r + td.rowSpan >= rows.length });
+            }
+        }
+        // ... then measure everything once ...
+        cells.forEach(function (it) {
+            var wantTop = declaredBorder(it.td, "Top");
+            var wantBottom = declaredBorder(it.td, "Bottom");
+            if (!(wantTop > 0) && !(wantBottom > 0)) return;
+            var cs = window.getComputedStyle(it.td);
+            it.padTop = parseFloat(cs.paddingTop) || 0;
+            it.padBottom = parseFloat(cs.paddingBottom) || 0;
+            it.dTop = wantTop > 0 ? wantTop - (parseFloat(cs.borderTopWidth) || 0) : 0;
+            it.dBottom = it.last && wantBottom > 0 ? wantBottom - (parseFloat(cs.borderBottomWidth) || 0) : 0;
+        });
+        // ... and write the differences
+        cells.forEach(function (it) {
+            if (it.dTop > 0.01) it.td.style.paddingTop = (it.padTop + it.dTop) + "px";
+            if (it.dBottom > 0.01) it.td.style.paddingBottom = (it.padBottom + it.dBottom) + "px";
+        });
+    }
+
+    /* Google Docs makes every table row a whole number of pixels tall
+       (rounding up), so a row of one 10pt line is 31px, not 30.67. Read all
+       the rows first and write after, so this costs one layout, not one
+       per row. */
+    function roundRows(tbl) {
+        var scale = scaleOf(tbl) || 1;
+        var rows = [];
+        for (var r = 0; r < tbl.rows.length; r++) {
+            var row = tbl.rows[r];
+            if (row.classList.contains("doc-autobreak") || row.closest("table") !== tbl) continue;
+            rows.push({ row: row, h: row.getBoundingClientRect().height / scale });
+        }
+        var writes = [];
+        rows.forEach(function (it) {
+            var extra = Math.ceil(it.h - 0.05) - it.h;
+            if (extra <= 0.01) return;
+            for (var c = 0; c < it.row.cells.length; c++) {
+                var td = it.row.cells[c];
+                if (td.rowSpan > 1) continue;
+                writes.push({ td: td, pb: (parseFloat(window.getComputedStyle(td).paddingBottom) || 0) + extra });
+            }
+        });
+        writes.forEach(function (w) { w.td.style.paddingBottom = w.pb + "px"; });
+    }
+    // the border width the document states (the specified value, before the
+    // browser rounds it to device pixels)
+    function declaredBorder(td, side) {
+        var st = td.style["border" + side + "Style"];
+        if (!st || st === "none" || st === "hidden") return 0;
+        return cssLenPx(td.style["border" + side + "Width"]);
+    }
+    function restoreTablePadding(root) {
+        var cells = root.querySelectorAll("[data-pad0]");
+        for (var i = 0; i < cells.length; i++) {
+            cells[i].style.padding = cells[i].getAttribute("data-pad0");
+            cells[i].removeAttribute("data-pad0");
+        }
+    }
+
+    /* ---------------- list numbering ---------------- */
+
+    var BULLETS = [String.fromCharCode(0x25CF), String.fromCharCode(0x25CB), String.fromCharCode(0x25A0)];
+    var ORDERED = ["decimal", "lowerLetter", "lowerRoman"];
+
+    function roman(n) {
+        if (n < 1 || n > 3999) return String(n);
+        var v = [1000, 900, 500, 400, 100, 90, 50, 40, 10, 9, 5, 4, 1];
+        var s = ["m", "cm", "d", "cd", "c", "xc", "l", "xl", "x", "ix", "v", "iv", "i"];
+        var out = "";
+        for (var i = 0; i < v.length; i++) while (n >= v[i]) { out += s[i]; n -= v[i]; }
+        return out;
+    }
+    function formatNumber(n, fmt) {
+        switch (fmt) {
+            case "lowerLetter":
+            case "upperLetter":
+                if (n < 1) return String(n);
+                var ch = String.fromCharCode(97 + (n - 1) % 26);
+                var s = new Array(Math.floor((n - 1) / 26) + 2).join(ch);
+                return fmt === "upperLetter" ? s.toUpperCase() : s;
+            case "lowerRoman": return roman(n);
+            case "upperRoman": return roman(n).toUpperCase();
+            case "decimalZero": return n < 10 ? "0" + n : String(n);
+            case "none":
+            case "bullet": return "";
+        }
+        return String(n);
+    }
+
+    function isList(el) { return el && (el.tagName === "OL" || el.tagName === "UL"); }
+
+    /* depth of a list among its list ancestors (0 = outermost) */
+    function listDepth(list, root) {
+        var d = 0;
+        for (var p = list.parentNode; p && p !== root; p = p.parentNode) {
+            if (isList(p)) d++;
+        }
+        return d;
+    }
+
+    function applyNumbering(root) {
+        if (!root) return;
+        var lists = root.querySelectorAll("ol, ul");
+        // continuation state per list instance and level
+        var carried = {};
+        // a list split across a page carries on in its tail, and the tail
+        // of a split item is the same item (no marker, no count)
+        var splitEnd = {}, splitItem = {};
+        for (var i = 0; i < lists.length; i++) {
+            var list = lists[i];
+            if (list.classList.contains("of-checklist")) continue;
+            var depth = listDepth(list, root);
+            var ordered = list.tagName === "OL";
+            var fmt = list.getAttribute("data-fmt") ||
+                (ordered ? ORDERED[depth % ORDERED.length] : "bullet");
+            var text = list.getAttribute("data-lvltext");
+            if (text === null) {
+                text = fmt === "bullet" ? BULLETS[depth % BULLETS.length] : "%" + (depth + 1) + ".";
+            }
+            var num = list.getAttribute("data-num");
+            var key = num ? num + ":" + depth : null;
+            var start = parseInt(list.getAttribute("start"), 10);
+            var counter;
+            var listOf = list.getAttribute("data-split-of");
+            if (key && carried[key] !== undefined) counter = carried[key];
+            else if (listOf && splitEnd[listOf] !== undefined) counter = splitEnd[listOf];
+            else counter = (isNaN(start) ? 1 : start) - 1;
+
+            for (var c = list.firstElementChild; c; c = c.nextElementSibling) {
+                if (c.tagName !== "LI") continue;
+                var itemOf = c.getAttribute("data-split-of");
+                if (itemOf) {
+                    var headItem = splitItem[itemOf];
+                    c.setAttribute("data-n", headItem ? headItem.getAttribute("data-n") : counter);
+                    if (c.getAttribute("data-marker") !== "") c.setAttribute("data-marker", "");
+                    if (c.hasAttribute("data-split")) splitItem[c.getAttribute("data-split")] = headItem || c;
+                    continue;
+                }
+                if (c.hasAttribute("data-split")) splitItem[c.getAttribute("data-split")] = c;
+                var v = parseInt(c.getAttribute("value"), 10);
+                counter = isNaN(v) ? counter + 1 : v;
+                c.setAttribute("data-n", counter);
+                var marker = text.replace(/%(\d)/g, function (m, lvl) {
+                    var want = parseInt(lvl, 10) - 1;
+                    if (want === depth) return formatNumber(counter, fmt);
+                    var anc = ancestorItem(c, want, root);
+                    if (!anc) return "";
+                    var ancList = anc.parentNode;
+                    var ancFmt = ancList.getAttribute("data-fmt") ||
+                        (ancList.tagName === "OL" ? ORDERED[want % ORDERED.length] : "bullet");
+                    return formatNumber(parseInt(anc.getAttribute("data-n"), 10) || 1, ancFmt);
+                });
+                if (fmt === "none") marker = text.replace(/%\d/g, "");
+                if (c.getAttribute("data-marker") !== marker) c.setAttribute("data-marker", marker);
+            }
+            if (key) carried[key] = counter;
+            if (list.hasAttribute("data-split")) splitEnd[list.getAttribute("data-split")] = counter;
+            // a deeper level restarts once a shallower item of the same list
+            // comes along
+            if (key) {
+                for (var k in carried) {
+                    var parts = k.split(":");
+                    if (parts[0] === num && parseInt(parts[1], 10) > depth) delete carried[k];
+                }
+            }
+        }
+    }
+
+    // the list item at a given depth that a nested list item hangs off
+    function ancestorItem(li, depth, root) {
+        var list = li.parentNode;
+        while (list && list !== root) {
+            var parent = list.parentNode;
+            var d = listDepth(list, root);
+            if (d === depth + 1 || (d > depth && isList(parent) && listDepth(parent, root) === depth)) {
+                // nested directly in a list: the item before it
+                if (isList(parent)) {
+                    for (var p = list.previousElementSibling; p; p = p.previousElementSibling) {
+                        if (p.tagName === "LI") return p;
+                    }
+                    return null;
+                }
+                if (parent && parent.tagName === "LI") return parent;
+            }
+            list = parent;
+        }
+        return null;
+    }
+
+    /* ---------------- tabs ---------------- */
+
+    // the edge tab stops are measured from: a table cell's content box, or
+    // the text column
+    function tabOrigin(span, rootEl) {
+        for (var p = span.parentNode; p && p !== rootEl; p = p.parentNode) {
+            if (p.tagName === "TD" || p.tagName === "TH") {
+                var r = p.getBoundingClientRect();
+                var cs = window.getComputedStyle(p);
+                return r.left + (parseFloat(cs.borderLeftWidth) || 0) + (parseFloat(cs.paddingLeft) || 0) * scaleOf(p);
+            }
+        }
+        var rr = rootEl.getBoundingClientRect();
+        var rcs = window.getComputedStyle(rootEl);
+        return rr.left + (parseFloat(rcs.paddingLeft) || 0) * scaleOf(rootEl);
+    }
+    function scaleOf(el) {
+        var h = el.offsetWidth;
+        var r = el.getBoundingClientRect().width;
+        return h > 0 && r > 0 ? r / h : 1;
+    }
+    function parseTabs(block) {
+        var out = [];
+        for (var b = block; b; b = b.parentElement) {
+            var raw = b.getAttribute && b.getAttribute("data-tabs");
+            if (raw) {
+                raw.split(";").forEach(function (t) {
+                    var p = t.split(":");
+                    var pos = parseFloat(p[1]);
+                    if (!isNaN(pos)) out.push({ align: p[0] || "left", pos: pos * PT, leader: p[2] || "none" });
+                });
+                break;
+            }
+            if (!isInlineEl(b) && !holdsInline(b)) break;
+        }
+        out.sort(function (a, b) { return a.pos - b.pos; });
+        return out;
+    }
+    function lineBlockOf(el, rootEl) {
+        for (var p = el.parentNode; p && p !== rootEl; p = p.parentNode) {
+            if (p.nodeType === 1 && !isInlineEl(p)) return p;
+        }
+        return rootEl;
+    }
+
+    function applyTabs(root) {
+        if (!root) return;
+        // a tab's position only depends on what comes before it, and the
+        // ones before it are sized first - so each tab can be measured as
+        // it stands and written only when its width really changes, which
+        // on an unchanged document means no layout work at all
+        var spans = root.querySelectorAll("span.doc-tab");
+        for (var i = 0; i < spans.length; i++) sizeTab(spans[i], root);
+    }
+
+    function sizeTab(span, root) {
+        var block = lineBlockOf(span, root);
+        var scale = scaleOf(root) || 1;
+        if (span.style.display !== "inline-block") span.style.display = "inline-block";
+        var origin = tabOrigin(span, root);
+        var x = (span.getBoundingClientRect().left - origin) / scale;
+        var stops = parseTabs(block);
+        var stop = null;
+        for (var i = 0; i < stops.length; i++) {
+            if (stops[i].pos > x + 0.5) { stop = stops[i]; break; }
+        }
+        var w;
+        if (!stop) {
+            var grid = DEFAULT_TAB_PT * PT;
+            w = (Math.floor(x / grid + 1e-6) + 1) * grid - x;
+            if (w < 1) w += grid;
+        } else if (stop.align === "right" || stop.align === "center" || stop.align === "decimal") {
+            // the text after the tab, up to the next tab or the end of the line
+            var follow = followingWidth(span, block) / scale;
+            // a right stop sits on the margin more often than not; a hair of
+            // slack keeps the number from wrapping to a line of its own
+            w = stop.pos - x - (stop.align === "center" ? follow / 2 : follow) - 1;
+            if (w < 0) w = 0;
+        } else {
+            w = stop.pos - x;
+        }
+        w = Math.max(0, w);
+        if (Math.abs((parseFloat(span.style.width) || -1) - w) > 0.25) span.style.width = w + "px";
+        var leader = stop && stop.leader && stop.leader !== "none" ? stop.leader : null;
+        if (span.getAttribute("data-leader") !== leader) {
+            if (leader) span.setAttribute("data-leader", leader);
+            else span.removeAttribute("data-leader");
+        }
+    }
+
+    function followingWidth(span, block) {
+        var range = document.createRange();
+        range.setStartAfter(span);
+        var end = null;
+        var walker = document.createTreeWalker(block, NodeFilter.SHOW_ELEMENT, null);
+        walker.currentNode = span;
+        var n;
+        while ((n = walker.nextNode())) {
+            if (n.classList && n.classList.contains("doc-tab")) { end = n; break; }
+            if (n.tagName === "BR") { end = n; break; }
+        }
+        if (end) range.setEndBefore(end);
+        else range.setEnd(block, block.childNodes.length);
+        var rects = range.getClientRects();
+        if (!rects.length) return 0;
+        // only what stays on the tab's own line counts
+        var top = span.getBoundingClientRect().top;
+        var left = Infinity, right = -Infinity;
+        for (var i = 0; i < rects.length; i++) {
+            if (Math.abs(rects[i].top - top) > rects[i].height) continue;
+            left = Math.min(left, rects[i].left);
+            right = Math.max(right, rects[i].right);
+        }
+        return right > left ? right - left : 0;
+    }
+
+    /* ---------------- footnote references ---------------- */
+
+    function numberFootnotes(root) {
+        var order = [];
+        if (!root) return order;
+        var refs = root.querySelectorAll("sup.doc-fnref");
+        for (var i = 0; i < refs.length; i++) {
+            var id = refs[i].getAttribute("data-fn");
+            var n = order.indexOf(id);
+            if (n < 0) { order.push(id); n = order.length - 1; }
+            var label = String(n + 1);
+            if (refs[i].textContent !== label) refs[i].textContent = label;
+            refs[i].setAttribute("contenteditable", "false");
+        }
+        return order;
+    }
+
+    /* ---------------- pagination ---------------- */
+
+    function Paginator(o) {
+        this.o = o;
+        this.editor = o.editor;
+        this.pageEl = o.pageEl;
+        this.sheetH = o.sheetH;
+        this.gap = o.gap;
+        this.mTop = o.mTop;
+        this.mBot = o.mBot;
+    }
+
+    // y of an element's border-box top in #page padding-box coordinates
+    Paginator.prototype.top = function (el) {
+        var y = 0;
+        var n = el;
+        while (n && n !== this.pageEl) {
+            y += n.offsetTop;
+            n = n.offsetParent;
+            if (n === document.body || !n) {
+                // #page is not the offset parent chain root: measure
+                return this.rectTop(el.getBoundingClientRect());
+            }
+        }
+        return y;
+    };
+    Paginator.prototype.bottom = function (el) {
+        return this.top(el) + el.offsetHeight;
+    };
+    Paginator.prototype.scale = function () {
+        var h = this.pageEl.offsetHeight;
+        var r = this.pageEl.getBoundingClientRect().height;
+        return h > 0 && r > 0 ? r / h : 1;
+    };
+    Paginator.prototype.rectTop = function (rect) {
+        var pr = this.pageEl.getBoundingClientRect();
+        return (rect.top - pr.top) / this._scale - this.pageEl.clientTop;
+    };
+    Paginator.prototype.rectBottom = function (rect) {
+        var pr = this.pageEl.getBoundingClientRect();
+        return (rect.bottom - pr.top) / this._scale - this.pageEl.clientTop;
+    };
+    Paginator.prototype.sheetTop = function (i) { return i * (this.sheetH + this.gap); };
+    Paginator.prototype.contentTop = function (i) { return this.sheetTop(i) + this.mTop; };
+    Paginator.prototype.contentBottom = function (i) { return this.sheetTop(i) + this.sheetH - this.mBot; };
+
+    /* ---- splitting what crosses a page ----
+
+       A page boundary is real in the DOM. Whatever runs across it - a
+       paragraph, the list or quote around it, a table row and the cells of
+       that row - is cut into two elements: the head keeps what fits on the
+       page, a shallow copy (the tail) receives the rest, and a spacer
+       between them pushes the tail to the top of the next sheet. A border,
+       a shading or a cell rule therefore ends at the bottom of its page and
+       starts again on the next one, and nothing is left painted across the
+       gap between the sheets.
+
+       The two halves are paired with a token: data-split on the head,
+       data-split-of on the tail, data-pair on the spacer. Undoing a split
+       (before a relayout, and on everything that is saved) moves the tail's
+       content back into its head - an inner pair (a span or a cell cut in
+       the same place) is joined along the way - and the text node that was
+       divided is joined again. The caret is followed through all of it. */
+
+    var splitSeq = 0;
+    var splitBase = Math.floor(Math.random() * 46656).toString(36) + "-";
+    function newToken() {
+        splitSeq++;
+        return splitBase + splitSeq.toString(36);
+    }
+
+    // the selection, followed through the node moves a split makes
+    var track = null;
+    function beginTrack(root) {
+        var outer = track;
+        var sel = window.getSelection ? window.getSelection() : null;
+        if (!outer && sel && sel.rangeCount && sel.anchorNode && root.contains(sel.anchorNode)) {
+            track = {
+                sel: sel, a: sel.anchorNode, ao: sel.anchorOffset,
+                f: sel.focusNode, fo: sel.focusOffset, moved: false
+            };
+        }
+        return { outer: outer, mine: !outer && !!track };
+    }
+    function endTrack(t) {
+        if (!t.mine) return;
+        var tr = track;
+        track = null;
+        if (!tr || !tr.moved) return;
+        if (!tr.a.isConnected || !tr.f.isConnected) return;
+        var len = function (n) { return n.nodeType === 3 ? n.nodeValue.length : n.childNodes.length; };
+        try {
+            tr.sel.setBaseAndExtent(tr.a, Math.min(tr.ao, len(tr.a)), tr.f, Math.min(tr.fo, len(tr.f)));
+        } catch (e) { /* a selection the browser will not take back: leave it */ }
+    }
+    function noteMoved() {
+        if (track) track.moved = true;
+    }
+    function splitText(node, offset) {
+        var tail = node.splitText(offset);
+        if (track) {
+            if (track.a === node && track.ao > offset) { track.a = tail; track.ao -= offset; }
+            if (track.f === node && track.fo > offset) { track.f = tail; track.fo -= offset; }
+            track.moved = true;
+        }
+        return tail;
+    }
+    function joinText(prev, next) {
+        var n = prev.nodeValue.length;
+        if (track) {
+            if (track.a === next) { track.a = prev; track.ao += n; }
+            if (track.f === next) { track.f = prev; track.fo += n; }
+            track.moved = true;
+        }
+        prev.nodeValue += next.nodeValue;
+        next.parentNode.removeChild(next);
+    }
+
+    var SPLIT_BOX = { TR: 1, TD: 1, TH: 1, TBODY: 1, THEAD: 1, TFOOT: 1 };
+    function cloneShell(el) {
+        var c = el.cloneNode(false);
+        c.removeAttribute("id");
+        c.removeAttribute("data-split");
+        c.removeAttribute("data-split-of");
+        c.removeAttribute("data-marker");
+        c.classList.remove("doc-split-head", "doc-split-tail");
+        if (!c.getAttribute("class")) c.removeAttribute("class");
+        if (el.tagName === "OL") c.removeAttribute("start");
+        if (el.tagName === "TR") {
+            c.style.height = "";
+            c.removeAttribute("height");
+            if (!c.getAttribute("style")) c.removeAttribute("style");
+        }
+        if (el.tagName === "TABLE") {
+            // the tail of a table needs the column widths too
+            for (var k = el.firstElementChild; k; k = k.nextElementSibling) {
+                if (k.tagName !== "COLGROUP") continue;
+                var cg = k.cloneNode(true);
+                cg.setAttribute("data-split-copy", "1");
+                c.appendChild(cg);
+            }
+        }
+        return c;
+    }
+    function pairUp(head, tail) {
+        var k = newToken();
+        head.setAttribute("data-split", k);
+        tail.setAttribute("data-split-of", k);
+        if (!SPLIT_BOX[head.tagName] && !isInlineEl(head)) {
+            head.classList.add("doc-split-head");
+            tail.classList.add("doc-split-tail");
+        }
+        return k;
+    }
+    /* unmark drops one role from an element - "head" or "tail" - or both.
+       An element can be both at once (the middle of a row that spans three
+       pages), so undoing one split must leave the other alone. deep strips
+       everything below it too (a split that is being given up). */
+    function unmark(el, role, deep) {
+        var list = deep ? [el].concat(Array.prototype.slice.call(el.querySelectorAll("[data-split],[data-split-of]"))) : [el];
+        list.forEach(function (n, i) {
+            var r = i === 0 ? role : "both";
+            if (r !== "tail") {
+                n.removeAttribute("data-split");
+                n.classList.remove("doc-split-head");
+            }
+            if (r !== "head") {
+                n.removeAttribute("data-split-of");
+                n.classList.remove("doc-split-tail");
+            }
+            if (!n.getAttribute("class")) n.removeAttribute("class");
+        });
+    }
+    function meaningful(n) {
+        if (n.nodeType === 3) return /\S/.test(n.nodeValue);
+        return n.nodeType === 1 && !isSpacer(n) && !n.hasAttribute("data-split-copy");
+    }
+    function prevMeaningful(n) {
+        for (var p = n.previousSibling; p; p = p.previousSibling) if (meaningful(p)) return p;
+        return null;
+    }
+    function nextMeaningful(n) {
+        for (var p = n.nextSibling; p; p = p.nextSibling) if (meaningful(p)) return p;
+        return null;
+    }
+    function nextInOrder(node, stop) {
+        for (var n = node; n && n !== stop; n = n.parentNode) {
+            if (n.nextSibling) return n.nextSibling;
+        }
+        return null;
+    }
+
+    /* splitTree moves `start` and everything after it, up to the end of
+       `boundary`, into shallow copies of the ancestors in between. Returns
+       the node directly inside boundary that the moved content begins with.
+       An ancestor whose content all moves is taken along whole instead of
+       being copied, so no half is ever left empty. */
+    function splitTree(start, boundary) {
+        var cur = start;
+        while (cur && cur.parentNode && cur.parentNode !== boundary) {
+            var parent = cur.parentNode;
+            if (!prevMeaningful(cur)) {
+                cur = parent;
+                continue;
+            }
+            var clone = cloneShell(parent);
+            pairUp(parent, clone);
+            parent.parentNode.insertBefore(clone, parent.nextSibling);
+            var n = cur;
+            while (n) {
+                var nx = n.nextSibling;
+                clone.appendChild(n);
+                n = nx;
+            }
+            noteMoved();
+            cur = clone;
+        }
+        return cur;
+    }
+
+    // the node a line cut starts the moved content with (text divided there)
+    function lineStartNode(cut) {
+        if (cut.beforeEl) return cut.beforeEl;
+        var node = cut.node;
+        var len = node.nodeValue.length;
+        var at = cut.offset;
+        // the space a line wrapped at stays at the end of the line above,
+        // where it collapses; at the start of the next page it would not
+        var ws = node.parentNode ? window.getComputedStyle(node.parentNode).whiteSpace : "normal";
+        if (ws.indexOf("pre") !== 0 && ws !== "break-spaces") {
+            while (at > 0 && at < len && /\s/.test(node.nodeValue.charAt(at))) at++;
+        }
+        cut = { el: cut.el, node: node, offset: at };
+        if (cut.offset > 0 && cut.offset < len) return splitText(node, cut.offset);
+        if (cut.offset >= len) return nextInOrder(node, cut.el);
+        return node;
+    }
+
+    /* mergePair undoes one split: the tail's content goes back to the end
+       of its head (for a table row, cell by cell) and the tail goes away */
+    function mergePair(head, tail) {
+        if (!head || !tail || !tail.parentNode) return;
+        noteMoved();
+        if (head.tagName === "TR" && tail.tagName === "TR") {
+            var tcells = Array.prototype.slice.call(tail.cells);
+            for (var i = 0; i < tcells.length; i++) {
+                var hc = head.cells[i];
+                if (hc) {
+                    mergeChildren(hc, tcells[i]);
+                    unmark(hc, "head");
+                } else {
+                    unmark(tcells[i], "tail");
+                    head.appendChild(tcells[i]);
+                }
+            }
+        } else {
+            mergeChildren(head, tail);
+        }
+        if (tail.parentNode) tail.parentNode.removeChild(tail);
+        unmark(head, "head");
+    }
+    function mergeChildren(head, tail) {
+        var c, nx;
+        for (c = tail.firstChild; c; c = nx) {
+            nx = c.nextSibling;
+            if (c.nodeType === 1 && c.hasAttribute("data-split-copy")) tail.removeChild(c);
+        }
+        // an inner pair cut at the same place (a span, a list, a nested
+        // table) joins first, so the content meets inside it
+        var last = head.lastChild;
+        while (last && !meaningful(last)) last = last.previousSibling;
+        var first = tail.firstChild;
+        while (first && !meaningful(first)) first = first.nextSibling;
+        if (last && first && last.nodeType === 1 && first.nodeType === 1) {
+            var k = first.getAttribute("data-split-of");
+            if (k && last.getAttribute("data-split") === k) mergePair(last, first);
+        }
+        var seam = head.lastChild;
+        while (tail.firstChild) head.appendChild(tail.firstChild);
+        if (seam && seam.nodeType === 3 && seam.nextSibling && seam.nextSibling.nodeType === 3) {
+            joinText(seam, seam.nextSibling);
+        }
+    }
+
+    // undo the split a spacer stands for (the spacer is already gone)
+    function mergeAround(root, k, prev, next) {
+        if (!k) return;
+        var head = prev && prev.nodeType === 1 && prev.getAttribute("data-split") === k ? prev : null;
+        var tail = next && next.nodeType === 1 && next.getAttribute("data-split-of") === k ? next : null;
+        if (!head) {
+            var hs = root.querySelectorAll('[data-split="' + k + '"]');
+            head = hs.length ? hs[hs.length - 1] : null;
+        }
+        if (!tail) tail = root.querySelector('[data-split-of="' + k + '"]');
+        if (head && tail) mergePair(head, tail);
+        else {
+            if (head) unmark(head, "head");
+            if (tail) unmark(tail, "tail");
+        }
+    }
+
+    function removeSpacerList(list, root) {
+        for (var i = list.length - 1; i >= 0; i--) {
+            var el = list[i];
+            var parent = el.parentNode;
+            if (!parent) continue;
+            var k = el.getAttribute("data-pair");
+            var prev = el.previousSibling, next = el.nextSibling;
+            while (prev && !meaningful(prev) && !isSpacer(prev)) prev = prev.previousSibling;
+            while (next && !meaningful(next) && !isSpacer(next)) next = next.nextSibling;
+            parent.removeChild(el);
+            noteMoved();
+            if (k) {
+                mergeAround(root, k, prev, next);
+            } else if (prev && next && prev.nodeType === 3 && next.nodeType === 3 &&
+                    prev.nextSibling === next) {
+                // a line spacer from an older layout divided this text
+                joinText(prev, next);
+            }
+        }
+    }
+
+    function isInner(el) {
+        var p = el.parentNode;
+        return !!(p && p.nodeType === 1 && (p.hasAttribute("data-split") || p.hasAttribute("data-split-of")));
+    }
+    /* repairSplits puts right what editing did to a split: a tail whose
+       spacer was deleted joins its head again, and a marker left without
+       its partner (the tail typed away, a copy made by Enter) is dropped.
+       With all, every remaining pair is undone - what is saved has none. */
+    function repairSplits(root, all) {
+        var tails = root.querySelectorAll("[data-split-of]");
+        var i;
+        for (i = tails.length - 1; i >= 0; i--) {
+            var tl = tails[i];
+            if (!tl.parentNode || !root.contains(tl) || isInner(tl)) continue;
+            var k = tl.getAttribute("data-split-of");
+            var prev = tl.previousSibling;
+            while (prev && !meaningful(prev) && !isSpacer(prev)) prev = prev.previousSibling;
+            if (!all && prev && isSpacer(prev) && prev.getAttribute("data-pair") === k) continue;
+            var head = prev && prev.nodeType === 1 && prev.getAttribute("data-split") === k ? prev : null;
+            if (head) mergePair(head, tl);
+            else unmark(tl, "tail", true);
+        }
+        var heads = root.querySelectorAll("[data-split]");
+        for (i = 0; i < heads.length; i++) {
+            var h = heads[i];
+            if (!h.isConnected || !h.hasAttribute("data-split") || isInner(h)) continue;
+            var nx = h.nextSibling;
+            while (nx && !meaningful(nx) && !isSpacer(nx)) nx = nx.nextSibling;
+            if (!all && nx && isSpacer(nx) && nx.getAttribute("data-pair") === h.getAttribute("data-split")) continue;
+            unmark(h, "head", true);
+        }
+    }
+
+    function removeSpacers(root) {
+        var t = beginTrack(root);
+        try {
+            removeSpacerList(root.querySelectorAll(".doc-autobreak"), root);
+            repairSplits(root, true);
+            var breaks = root.querySelectorAll(".doc-pagebreak");
+            for (var i = 0; i < breaks.length; i++) breaks[i].style.height = "0px";
+        } finally {
+            endTrack(t);
+        }
+    }
+
+    /* unsplitWithin undoes the splits inside an element (a table about to
+       gain or lose a row or column) */
+    function unsplitWithin(root, el) {
+        if (!root || !el) return false;
+        var list = el.querySelectorAll(".doc-autobreak");
+        if (!list.length) return false;
+        var t = beginTrack(root);
+        try {
+            removeSpacerList(list, root);
+            repairSplits(root, false);
+        } finally {
+            endTrack(t);
+        }
+        return true;
+    }
+
+    /* unsplitAtCaret is called before a key edits the document. A
+       Backspace at the start of what a page break moved, a Delete at the
+       end of what it left behind, or any key over a selection that spans a
+       page boundary would otherwise act on the spacer rather than on the
+       text; the split in the way is undone first, and the edit then does
+       what it would do in one continuous paragraph. */
+    function edgeEmpty(el, caretNode, caretOffset, atStart) {
+        var r = document.createRange();
+        try {
+            if (atStart) {
+                r.setStart(el, 0);
+                r.setEnd(caretNode, caretOffset);
+            } else {
+                r.setStart(caretNode, caretOffset);
+                r.setEnd(el, el.childNodes.length);
+            }
+        } catch (e) { return false; }
+        if (r.toString().replace(/\s/g, "").split(String.fromCharCode(0x200B)).join("") !== "") return false;
+        var frag = r.cloneContents();
+        return !frag.querySelector || !frag.querySelector("img,table,hr");
+    }
+    function unsplitAtCaret(root, backward) {
+        var sel = window.getSelection ? window.getSelection() : null;
+        if (!sel || !sel.rangeCount) return false;
+        var r = sel.getRangeAt(0);
+        if (!root.contains(r.commonAncestorContainer)) return false;
+        var spacers = [];
+        var all, i;
+        if (!r.collapsed) {
+            all = root.querySelectorAll(".doc-autobreak");
+            for (i = 0; i < all.length; i++) {
+                if (r.intersectsNode(all[i])) spacers.push(all[i]);
+            }
+        } else {
+            var node = backward ? r.startContainer : r.endContainer;
+            var off = backward ? r.startOffset : r.endOffset;
+            var el = node.nodeType === 1 ? node : node.parentNode;
+            while (el && el !== root) {
+                if (el.tagName !== "TR" && !edgeEmpty(el, node, off, backward) &&
+                        !((el.tagName === "TD" || el.tagName === "TH") && el.hasAttribute(backward ? "data-split-of" : "data-split"))) {
+                    break;
+                }
+                var sib = backward ? el.previousSibling : el.nextSibling;
+                while (sib && !meaningful(sib) && !isSpacer(sib)) sib = backward ? sib.previousSibling : sib.nextSibling;
+                if (sib && isSpacer(sib)) {
+                    spacers.push(sib);
+                    break;
+                }
+                if (el.tagName === "TD" || el.tagName === "TH") {
+                    el = el.parentNode;
+                    continue;
+                }
+                if (sib) break;
+                el = el.parentNode;
+            }
+        }
+        if (!spacers.length) return false;
+        var t = beginTrack(root);
+        try {
+            removeSpacerList(spacers, root);
+            repairSplits(root, false);
+        } finally {
+            endTrack(t);
+        }
+        return true;
+    }
+
+    function blockKids(container) {
+        var out = [];
+        for (var c = container.firstElementChild; c; c = c.nextElementSibling) {
+            if (isSpacer(c)) continue;
+            if (c.tagName === "COLGROUP" || c.tagName === "COL") continue;
+            out.push(c);
+        }
+        return out;
+    }
+
+    function keepsWithNext(el) {
+        if (el.getAttribute("data-keep-next") === "1") return true;
+        // headings written in the editor keep with what follows, like the
+        // Heading styles of every word processor
+        return /^H[1-6]$/.test(el.tagName) && !el.hasAttribute("data-keep-next");
+    }
+
+    /* A cut says where page content stops:
+         { kind: "before", el }        content from el moves (a block, li, row)
+         { kind: "line", el, node, offset | beforeEl }  a block splits at a line
+         { kind: "row", tr, cells: [cut|null per cell] }
+         { kind: "explicit", el }      a manual page break
+         { kind: "overflow" }          nothing movable: content spills over
+       with y = where the moved content currently starts, and boundary = the
+       element the split reaches up to (the editor, or a table cell). */
+
+    Paginator.prototype.findCut = function (container, B, C, boundary) {
+        var kids = blockKids(container);
+        if (!kids.length) return null;
+        // explicit breaks and page-break-before on this level
+        for (var e = 0; e < kids.length; e++) {
+            var k = kids[e];
+            if (k.classList.contains("doc-pagebreak")) {
+                var ty = this.top(k);
+                if (ty >= C - 1 && ty <= B + 0.5) return { kind: "explicit", el: k, y: ty, boundary: boundary };
+            } else if (k.getAttribute("data-page-break-before") === "1") {
+                var py = this.top(k);
+                if (py > C + 1 && py <= B + 0.5) return this.beforeCut(k, C, boundary);
+            }
+        }
+        // first child whose bottom crosses B (children are in flow order)
+        var lo = 0, hi = kids.length - 1, idx = -1;
+        while (lo <= hi) {
+            var mid = (lo + hi) >> 1;
+            if (this.bottom(kids[mid]) > B + 0.5) { idx = mid; hi = mid - 1; }
+            else lo = mid + 1;
+        }
+        if (idx < 0) return null;
+        // a float or a negative margin can leave an earlier child lower;
+        // walk back over anything that also crosses
+        while (idx > 0 && this.bottom(kids[idx - 1]) > B + 0.5) idx--;
+        // a child that only hangs its spacing-after over the edge still
+        // fits - the cut belongs to whatever comes after it
+        for (; idx < kids.length; idx++) {
+            if (idx > 0 && this.bottom(kids[idx]) <= B + 0.5) continue;
+            var cut = this.splitChild(kids[idx], B, C, boundary);
+            if (cut && cut.kind !== "fit") return cut;
+        }
+        return null;
+    };
+
+    Paginator.prototype.splitChild = function (el, B, C, boundary) {
+        var top = this.top(el);
+        if (top >= B - 0.5) return this.beforeCut(el, C, boundary);
+        var tag = el.tagName;
+        if (tag === "TABLE") return this.splitTable(el, B, C, boundary);
+        if (tag === "IMG" || tag === "HR" || tag === "VIDEO" || tag === "IFRAME") {
+            return this.beforeCut(el, C, boundary);
+        }
+        if (el.classList.contains("doc-pagebreak")) return null;
+        if (isList(el) || tag === "TBODY") {
+            var inner = this.findCut(el, B, C, boundary);
+            return this.hoist(inner, el, C, boundary);
+        }
+        if (hasBlockChild(el)) {
+            if (el.getAttribute("data-keep-lines") === "1" && top > C + 1) return this.beforeCut(el, C, boundary);
+            var sub = this.findCut(el, B, C, boundary);
+            return this.hoist(sub, el, C, boundary);
+        }
+        return this.splitLines(el, B, C, boundary);
+    };
+
+    // a cut before the first child of a container is a cut before the
+    // container (so keep-with-next and page-top checks see the real block)
+    Paginator.prototype.hoist = function (cut, container, C, boundary) {
+        if (!cut) return { kind: "fit" };
+        if (cut.kind === "before" && container !== boundary) {
+            var kids = blockKids(container);
+            if (kids.length && kids[0] === cut.el && this.top(container) > C + 1) {
+                return this.beforeCut(container, C, boundary);
+            }
+        }
+        return cut;
+    };
+
+    Paginator.prototype.beforeCut = function (el, C, boundary) {
+        var target = el;
+        // keep-with-next: pull the blocks that must stay with el along
+        for (var guard = 0; guard < 20; guard++) {
+            var prev = target.previousElementSibling;
+            while (prev && isSpacer(prev)) prev = prev.previousElementSibling;
+            if (!prev || prev.classList.contains("doc-pagebreak")) break;
+            if (!keepsWithNext(prev)) break;
+            if (this.top(prev) <= C + 1) break;
+            target = prev;
+        }
+        var y = this.top(target);
+        if (y <= C + 1) {
+            if (target !== el) {
+                target = el;
+                y = this.top(el);
+            }
+            if (y <= C + 1) return { kind: "overflow", y: y };
+        }
+        // a cut before the first row of a table is a cut before the table
+        if (target.tagName === "TR") {
+            var tbl = target.closest("table");
+            var rows = rowsOf(tbl);
+            if (rows[0] === target && tbl !== boundary) return this.beforeCut(tbl, C, boundary);
+        }
+        return { kind: "before", el: target, y: y, boundary: boundary };
+    };
+
+    function rowsOf(tbl) {
+        var out = [];
+        for (var i = 0; i < tbl.rows.length; i++) {
+            if (!tbl.rows[i].classList.contains("doc-autobreak") &&
+                tbl.rows[i].closest("table") === tbl) out.push(tbl.rows[i]);
+        }
+        return out;
+    }
+
+    Paginator.prototype.splitTable = function (tbl, B, C, boundary) {
+        var rows = rowsOf(tbl);
+        var idx = -1;
+        for (var i = 0; i < rows.length; i++) {
+            if (this.bottom(rows[i]) > B + 0.5) { idx = i; break; }
+        }
+        if (idx < 0) return { kind: "fit" };
+        var row = rows[idx];
+        var rowTop = this.top(row);
+        if (rowTop >= B - 1) return this.beforeCut(idx === 0 ? tbl : row, C, boundary);
+        // a row that cannot split, or one a merged cell reaches into from
+        // above, moves whole - unless it already starts the page
+        var spanned = rowspanCrosses(tbl, rows, idx);
+        var cant = row.getAttribute("data-cant-split") === "1" || spanned || hasRowspan(row);
+        if (cant && rowTop > C + 1) {
+            var bc = this.beforeCut(idx === 0 ? tbl : row, C, boundary);
+            if (bc.kind !== "overflow") return bc;
+        }
+        if (spanned || hasRowspan(row)) return { kind: "overflow", y: rowTop };
+        var cells = [];
+        var any = false, allNothing = true;
+        for (var c = 0; c < row.cells.length; c++) {
+            var td = row.cells[c];
+            var cs = window.getComputedStyle(td);
+            var padB = (parseFloat(cs.paddingBottom) || 0) + (parseFloat(cs.borderBottomWidth) || 0);
+            var cut = hasBlockChild(td) ? this.findCut(td, B - padB, C, td)
+                : this.splitLines(td, B - padB, C, td);
+            if (cut && cut.kind === "fit") cut = null;
+            if (cut && cut.kind === "overflow") {
+                // this cell cannot give anything up: move the row if we can
+                cut = { kind: "cellstart", td: td, y: rowTop };
+            }
+            if (cut) {
+                any = true;
+                if (cut.kind !== "cellstart" && !(cut.kind === "before" && cut.el === blockKids(td)[0])) {
+                    allNothing = false;
+                }
+            } else {
+                allNothing = false;
+            }
+            cells.push(cut);
+        }
+        if (!any) return { kind: "fit" };
+        if (allNothing && rowTop > C + 1) return this.beforeCut(idx === 0 ? tbl : row, C, boundary);
+        var y = Infinity;
+        cells.forEach(function (ct) { if (ct && ct.y < y) y = ct.y; });
+        return { kind: "row", tr: row, cells: cells, y: y, boundary: boundary };
+    };
+
+    function hasRowspan(row) {
+        for (var i = 0; i < row.cells.length; i++) {
+            if (row.cells[i].rowSpan > 1) return true;
+        }
+        return false;
+    }
+    function rowspanCrosses(tbl, rows, idx) {
+        for (var r = 0; r < idx; r++) {
+            for (var c = 0; c < rows[r].cells.length; c++) {
+                if (r + rows[r].cells[c].rowSpan > idx) return true;
+            }
+        }
+        return false;
+    }
+
+    /* the visual lines of a block that holds only inline content */
+    Paginator.prototype.lines = function (block) {
+        var frags = [];
+        var self = this;
+        var range = document.createRange();
+        function visit(parent) {
+            for (var n = parent.firstChild; n; n = n.nextSibling) {
+                if (n.nodeType === 3) {
+                    if (!n.nodeValue) continue;
+                    range.selectNodeContents(n);
+                    var rects = range.getClientRects();
+                    for (var i = 0; i < rects.length; i++) {
+                        if (rects[i].height <= 0) continue;
+                        frags.push({ node: n, index: i, top: self.rectTop(rects[i]), bottom: self.rectBottom(rects[i]) });
+                    }
+                } else if (n.nodeType === 1) {
+                    if (isSpacer(n)) continue;
+                    var disp = n.tagName === "IMG" ? "inline" : window.getComputedStyle(n).display;
+                    if (n.tagName === "IMG" || (disp === "inline-block" && !n.classList.contains("doc-tab"))) {
+                        // a picture or an inline-block is one unbreakable box
+                        var r = n.getBoundingClientRect();
+                        if (r.height > 0) frags.push({ el: n, top: self.rectTop(r), bottom: self.rectBottom(r) });
+                    } else if (disp !== "none") {
+                        visit(n);
+                    }
+                }
+            }
+        }
+        visit(block);
+        var lines = [];
+        frags.forEach(function (f) {
+            var cur = lines[lines.length - 1];
+            if (!cur || f.top >= cur.bottom - 1.5) {
+                lines.push({ top: f.top, bottom: f.bottom, first: f, text: !f.el });
+            } else {
+                cur.top = Math.min(cur.top, f.top);
+                cur.bottom = Math.max(cur.bottom, f.bottom);
+                if (!f.el) cur.text = true;
+            }
+        });
+        // a text rect is the glyphs, not the line: the line box reaches half
+        // the leading further down, and that is what has to fit the page
+        var lh = parseFloat(window.getComputedStyle(block).lineHeight) || 0;
+        lines.forEach(function (ln) {
+            var glyphs = ln.bottom - ln.top;
+            ln.boxBottom = ln.bottom + (ln.text && lh > glyphs ? (lh - glyphs) / 2 : 0);
+        });
+        return lines;
+    };
+
+    Paginator.prototype.splitLines = function (block, B, C, boundary) {
+        var lines = this.lines(block);
+        if (!lines.length) {
+            return this.bottom(block) > B + 0.5 && this.top(block) > C + 1 ?
+                this.beforeCut(block, C, boundary) : { kind: "fit" };
+        }
+        var k = -1;
+        for (var i = 0; i < lines.length; i++) {
+            if (lines[i].boxBottom > B + 0.5) { k = i; break; }
+        }
+        // only the spacing after the last line hangs over: that is allowed
+        if (k < 0) return { kind: "fit" };
+        var atTop = this.top(block) <= C + 1;
+        if (block.getAttribute("data-keep-lines") === "1" && !atTop) k = 0;
+        // widow/orphan control is on unless the paragraph turns it off
+        if (block.getAttribute("data-widow") !== "0" && lines.length > 1) {
+            // a widow (last line alone on the next page) pulls one more line
+            // over; an orphan (first line alone on this page) moves the
+            // whole paragraph
+            if (k > 0 && lines.length - k === 1) k = k - 1;
+            if (k === 1 && !atTop) k = 0;
+        }
+        if (k === 0) {
+            if (!atTop) return this.beforeCut(block, C, boundary);
+            if (lines.length === 1) return { kind: "overflow", y: lines[0].top };
+            k = 1;
+        }
+        var f = lines[k].first;
+        var cut = { kind: "line", el: block, y: lines[k].top, boundary: boundary };
+        if (f.el) {
+            cut.beforeEl = f.el;
+        } else {
+            cut.node = f.node;
+            cut.offset = f.index === 0 ? 0 : this.lineStartOffset(f.node, lines[k].top);
+        }
+        return cut;
+    };
+
+    // first character of a text node that sits on the line starting at y
+    Paginator.prototype.lineStartOffset = function (node, lineTop) {
+        var len = node.nodeValue.length;
+        var range = document.createRange();
+        var self = this;
+        function topAt(o) {
+            for (var i = o; i < len; i++) {
+                range.setStart(node, i);
+                range.setEnd(node, i + 1);
+                var rr = range.getClientRects();
+                if (rr.length && rr[0].height > 0) return { top: self.rectTop(rr[rr.length - 1]), at: i };
+            }
+            return null;
+        }
+        var lo = 0, hi = len - 1, ans = len;
+        while (lo <= hi) {
+            var mid = (lo + hi) >> 1;
+            var t = topAt(mid);
+            if (!t) { hi = mid - 1; continue; }
+            if (t.top >= lineTop - 1.5) { ans = Math.min(ans, t.at); hi = mid - 1; }
+            else lo = t.at + 1;
+        }
+        return Math.min(ans, len);
+    };
+
+    /* ---- applying a cut ---- */
+
+    function makeSpacer(kind, cols) {
+        var el;
+        if (kind === "row") {
+            el = document.createElement("tr");
+            var td = document.createElement("td");
+            td.colSpan = Math.max(1, cols);
+            td.className = "doc-autobreak-cell";
+            el.appendChild(td);
+        } else {
+            el = document.createElement("div");
+        }
+        el.className = "doc-autobreak";
+        el.setAttribute("contenteditable", "false");
+        el.setAttribute("aria-hidden", "true");
+        return el;
+    }
+
+    function marginTopOf(el) {
+        return parseFloat(window.getComputedStyle(el).marginTop) || 0;
+    }
+
+    // stretch a spacer so that the content after it starts at targetY
+    Paginator.prototype.land = function (spacer, targetY, probe) {
+        var box = spacer.tagName === "TR" ? spacer.firstChild : spacer;
+        box.style.height = "0px";
+        for (var i = 0; i < 3; i++) {
+            var at = probe ? probe() : this.bottom(spacer);
+            var cur = parseFloat(box.style.height) || 0;
+            var want = Math.max(0, cur + (targetY - at));
+            if (Math.abs(want - cur) < 0.05) break;
+            box.style.height = want + "px";
+        }
+    };
+
+    Paginator.prototype.apply = function (cut, nextC) {
+        var self = this;
+        var boundary = cut.boundary || this.editor;
+        var top, dv;
+        switch (cut.kind) {
+            case "explicit":
+                cut.el.style.height = "0px";
+                this.land(cut.el, nextC);
+                return;
+            case "before":
+                var el = cut.el;
+                if (el.tagName === "TR") {
+                    var cols = 0;
+                    for (var c = 0; c < el.cells.length; c++) cols += el.cells[c].colSpan;
+                    var sp = makeSpacer("row", cols);
+                    el.parentNode.insertBefore(sp, el);
+                    var bw = parseFloat(window.getComputedStyle(el.cells[0] || el).borderTopWidth) || 0;
+                    this.land(sp, nextC + bw / 2, function () { return self.top(el); });
+                    return;
+                }
+                // what holds el (a list, a quote) splits with it, so the
+                // spacer sits between two whole boxes
+                top = el.parentNode === boundary ? el : splitTree(el, boundary);
+                dv = makeSpacer("block");
+                if (top !== el && top.getAttribute("data-split-of")) dv.setAttribute("data-pair", top.getAttribute("data-split-of"));
+                top.parentNode.insertBefore(dv, top);
+                // spacing-before still applies at the top of a page
+                this.land(dv, nextC + marginTopOf(el), function () { return self.top(el); });
+                return;
+            case "line":
+                top = splitTree(lineStartNode(cut), boundary);
+                if (!top) return;
+                dv = makeSpacer("block");
+                if (top.getAttribute && top.getAttribute("data-split-of")) dv.setAttribute("data-pair", top.getAttribute("data-split-of"));
+                top.parentNode.insertBefore(dv, top);
+                this.land(dv, nextC, function () { return self.top(top); });
+                return;
+            case "row":
+                this.splitRow(cut, nextC, true);
+                return;
+        }
+    };
+
+    /* A row that breaks across the page becomes two rows: the head keeps
+       what fits in every cell, a copy of the row takes the rest of each
+       cell, and a spacer row between them carries the copy to the next
+       sheet. Each half has its own cell borders and shading. */
+    Paginator.prototype.splitRow = function (cut, nextC, withSpacer) {
+        var self = this;
+        var tr = cut.tr;
+        var tailTr = cloneShell(tr);
+        var k = pairUp(tr, tailTr);
+        tr.parentNode.insertBefore(tailTr, tr.nextSibling);
+        noteMoved();
+        var cells = Array.prototype.slice.call(tr.cells);
+        cells.forEach(function (td, i) {
+            var tailTd = cloneShell(td);
+            pairUp(td, tailTd);
+            tailTr.appendChild(tailTd);
+            var ct = cut.cells[i];
+            if (!ct) return;
+            var start = null;
+            if (ct.kind === "cellstart") {
+                start = td.firstChild;
+            } else if (ct.kind === "line") {
+                start = lineStartNode(ct);
+            } else if (ct.kind === "before" || ct.kind === "explicit") {
+                start = ct.el;
+            } else if (ct.kind === "row") {
+                // a nested table splits in the same place; its tail rows go
+                // with the rest of this cell
+                start = self.splitRow(ct, nextC, false);
+            }
+            if (!start) return;
+            var from = start.parentNode === td ? start : splitTree(start, td);
+            while (from) {
+                var nx = from.nextSibling;
+                tailTd.appendChild(from);
+                from = nx;
+            }
+        });
+        if (withSpacer) {
+            var cols = 0;
+            for (var c = 0; c < cells.length; c++) cols += cells[c].colSpan;
+            var sp = makeSpacer("row", cols);
+            sp.setAttribute("data-pair", k);
+            tr.parentNode.insertBefore(sp, tailTr);
+            var bw = parseFloat(window.getComputedStyle(tailTr.cells[0] || tailTr).borderTopWidth) || 0;
+            this.land(sp, nextC + bw / 2, function () { return self.top(tailTr); });
+        }
+        return tailTr;
+    };
+
+    /* ---- footnotes ---- */
+
+    Paginator.prototype.refsBetween = function (y0, y1) {
+        var out = [];
+        var refs = this.o.fnRefs || [];
+        for (var i = 0; i < refs.length; i++) {
+            var r = refs[i];
+            var rect = r.getBoundingClientRect();
+            if (!rect.height) continue;
+            var y = this.rectTop(rect);
+            if (y >= y0 - 1 && y < y1) {
+                var id = r.getAttribute("data-fn");
+                if (out.indexOf(id) < 0) out.push(id);
+            }
+        }
+        return out;
+    };
+
+    /* ---- the page loop ---- */
+
+    /* Where to start. An edit cannot move a page break that comes before
+       it, so when the caller says where the change is (fromY) and hands in
+       the previous page records, the pages up to the one before the change
+       are kept as they are - with their spacers and splits - and only the
+       rest is laid out again. Starting one page early leaves room for a
+       widow or a keep-with-next that pulls a line back across the page
+       above. */
+    Paginator.prototype.resume = function () {
+        var o = this.o;
+        var prev = o.prevPages;
+        if (!(o.fromY >= 0) || !prev || prev.length < 3) return 0;
+        var p = 0;
+        for (var k = 0; k < prev.length; k++) {
+            if (prev[k].sheetTop <= o.fromY) p = k;
+        }
+        var start = p - 1;
+        if (start < 1) return 0;
+        var threshold = this.contentTop(start);
+        var self = this;
+        var stale = [];
+        var spacers = this.editor.querySelectorAll(".doc-autobreak");
+        for (var i = 0; i < spacers.length; i++) {
+            if (self.top(spacers[i]) >= threshold - 0.5) stale.push(spacers[i]);
+        }
+        var breaks = this.editor.querySelectorAll(".doc-pagebreak");
+        var staleBreaks = [];
+        for (i = 0; i < breaks.length; i++) {
+            if (self.top(breaks[i]) >= threshold - 0.5) staleBreaks.push(breaks[i]);
+        }
+        removeSpacerList(stale, this.editor);
+        repairSplits(this.editor, false);
+        staleBreaks.forEach(function (b) { b.style.height = "0px"; });
+        return start;
+    };
+
+    Paginator.prototype.run = function () {
+        var o = this.o;
+        this._scale = this.scale();
+        var i = this.resume();
+        var pages;
+        if (i > 0) {
+            pages = o.prevPages.slice(0, i).map(function (pg) {
+                var copy = {};
+                for (var k in pg) copy[k] = pg[k];
+                return copy;
+            });
+        } else {
+            removeSpacers(this.editor);
+            pages = [];
+        }
+        var contentEnd = function (self) { return self.bottom(self.editor); };
+        for (var guard = 0; guard < 3000; guard++) {
+            this._scale = this.scale();
+            var C = this.contentTop(i);
+            var Bfull = this.contentBottom(i);
+            var B = Bfull;
+            var ids = [];
+            var fnH = 0;
+            var cut = null;
+            for (var iter = 0; iter < 4; iter++) {
+                cut = this.findCut(this.editor, B, C, this.editor);
+                if (cut && cut.kind === "fit") cut = null;
+                var endY = cut ? cut.y : contentEnd(this);
+                var got = o.measureFootnotes ? this.refsBetween(C, Math.min(endY, B + 0.5)) : [];
+                var h = got.length ? o.measureFootnotes(got) : 0;
+                if (got.join(",") === ids.join(",") && Math.abs(h - fnH) < 0.5) break;
+                ids = got;
+                fnH = h;
+                B = Bfull - fnH;
+            }
+            var page = {
+                index: i, sheetTop: this.sheetTop(i), contentTop: C, contentBottom: B,
+                footnotes: ids, footnoteTop: Bfull - fnH
+            };
+            pages.push(page);
+            if (!cut) break;
+            var nextC = this.contentTop(i + 1);
+            if (cut.kind !== "overflow") this.apply(cut, nextC);
+            i++;
+        }
+        return pages;
+    };
+
+    function paginate(opts) {
+        var t = beginTrack(opts.editor);
+        try {
+            return new Paginator(opts).run();
+        } finally {
+            endTrack(t);
+        }
+    }
+
+    return {
+        PT: PT,
+        MM: MM,
+        fontRatios: fontRatios,
+        lineHeightPx: lineHeightPx,
+        applyLineHeights: applyLineHeights,
+        applyTableBorders: applyTableBorders,
+        restoreTablePadding: restoreTablePadding,
+        restorePictureLines: restorePictureLines,
+        applyNumbering: applyNumbering,
+        applyTabs: applyTabs,
+        numberFootnotes: numberFootnotes,
+        removeSpacers: removeSpacers,
+        unsplitWithin: unsplitWithin,
+        unsplitAtCaret: unsplitAtCaret,
+        paginate: paginate,
+        formatNumber: formatNumber
+    };
+})();

+ 669 - 0
src/web/Office/docs/docs_pdf.js

@@ -0,0 +1,669 @@
+/*
+    ArozOS Office - Docs PDF export
+    ===============================
+    The PDF is drawn from the editor's own pages. docs_layout.js has already
+    decided where every line, row and page break goes, and the sheet the
+    editor shows is the sheet that gets written: each element in the page
+    becomes the PDF object it is, at the position the browser gave it.
+
+      text        real PDF text, one show-text per line fragment at the
+                  browser's baseline (OfficePdfCore.drawRuns - the font rules
+                  are the ones Slides uses: standard PDF fonts for families
+                  metric-identical to them, the shipped Noto faces embedded
+                  and subset for everything else, a picture of the run only
+                  when nothing can show a character)
+      background  a filled rectangle - per line fragment for highlighted text
+      borders     filled strips; a collapsed table rule is centred on the
+                  cell edge the way the browser draws it, at the width the
+                  document states rather than the pixel it was rounded to
+      pictures    the original bytes embedded once, cropped with a clip path
+      markers     list numbers and bullets (li[data-marker]), tab leaders and
+                  the footnote rule, which the editor draws with CSS
+                  generated content and therefore have no text node
+
+    Nothing is re-laid out, so the export cannot paginate differently from
+    the editor. The page is only held while it is measured: what to draw is
+    written down as a display list, and common/pdfworker.js (pdfdraw.js)
+    assembles the file in a Web Worker while the document stays editable.
+
+    Usage:
+        DocsPdf.build({ pageEl, pages, sheetW, sheetH, title,
+                        enter, leave, onProgress }) -> Promise<Uint8Array>
+*/
+
+var DocsPdf = (function () {
+    "use strict";
+
+    var C = OfficePdfCore;
+    var parseFill = C.parseFill, parseColor = C.parseColor;
+    var PT = 96 / 72;
+
+    // subtrees that are editor chrome, not page content
+    var SKIP_SELECTOR = "#pageSheets, #pageGuides, .doc-fn-measure, .of-img-handle, .doc-autobreak";
+
+    function skipped(el) {
+        return !!(el && el.closest && el.closest(SKIP_SELECTOR));
+    }
+
+    /* ---------------- geometry ---------------- */
+
+    function Frame(pageEl) {
+        this.pageEl = pageEl;
+        var r = pageEl.getBoundingClientRect();
+        this.left = r.left + pageEl.clientLeft;
+        this.top = r.top + pageEl.clientTop;
+    }
+    // a client rect in #page layout coordinates (px)
+    Frame.prototype.box = function (rect) {
+        return {
+            x: rect.left - this.left, y: rect.top - this.top,
+            w: rect.right - rect.left, h: rect.bottom - rect.top
+        };
+    };
+
+    function pageIndexOf(pages, sheetH, y) {
+        // pages are in order; the last sheet whose top is at or above y
+        var lo = 0, hi = pages.length - 1, ans = 0;
+        while (lo <= hi) {
+            var mid = (lo + hi) >> 1;
+            if (pages[mid].sheetTop <= y + 0.5) { ans = mid; lo = mid + 1; }
+            else hi = mid - 1;
+        }
+        return ans;
+    }
+    // every page a vertical band [y, y+h] touches
+    function pagesTouched(pages, sheetH, y, h) {
+        var out = [];
+        var i = pageIndexOf(pages, sheetH, y);
+        for (; i < pages.length; i++) {
+            var top = pages[i].sheetTop;
+            if (top > y + h) break;
+            if (top + sheetH > y) out.push(i);
+        }
+        return out;
+    }
+
+    /* ---------------- collecting what is on the pages ---------------- */
+
+    function collect(o, frame) {
+        var pages = o.pages, sheetH = o.sheetH;
+        var perPage = pages.map(function () {
+            return { fills: [], images: [], borders: [], texts: [], extras: [] };
+        });
+        var add = function (kind, y, h, item) {
+            pagesTouched(pages, sheetH, y, h).forEach(function (i) { perPage[i][kind].push(item); });
+        };
+
+        var all = o.pageEl.getElementsByTagName("*");
+        for (var i = 0; i < all.length; i++) {
+            var el = all[i];
+            if (skipped(el)) continue;
+            var cs = window.getComputedStyle(el);
+            if (cs.display === "none" || cs.visibility === "hidden") continue;
+            if (el.closest("[hidden]")) continue;
+            var rects;
+
+            // backgrounds
+            var bg = parseFill(cs.backgroundColor);
+            if (bg && el !== o.pageEl) {
+                rects = cs.display === "inline" ? el.getClientRects() : [el.getBoundingClientRect()];
+                for (var r = 0; r < rects.length; r++) {
+                    var b = frame.box(rects[r]);
+                    if (b.w <= 0 || b.h <= 0) continue;
+                    add("fills", b.y, b.h, { box: b, fill: bg });
+                }
+            }
+
+            // borders
+            if (cs.display !== "inline") {
+                var bb = frame.box(el.getBoundingClientRect());
+                var cell = (el.tagName === "TD" || el.tagName === "TH") && cs.borderCollapse === "collapse";
+                ["Top", "Right", "Bottom", "Left"].forEach(function (side) {
+                    var st = cs["border" + side + "Style"];
+                    var w = parseFloat(cs["border" + side + "Width"]) || 0;
+                    if (!w || st === "none" || st === "hidden") return;
+                    var col = parseFill(cs["border" + side + "Color"]);
+                    if (!col) return;
+                    // the width the document states, before pixel rounding
+                    var declared = el.style["border" + side + "Width"];
+                    if (declared && /pt$/.test(declared)) w = parseFloat(declared) * PT;
+                    add("borders", bb.y - w, bb.h + 2 * w, { box: bb, side: side, w: w, fill: col, centred: cell, dash: st });
+                });
+            }
+
+            // pictures
+            if (el.tagName === "IMG") {
+                var ib = frame.box(el.getBoundingClientRect());
+                var bl = parseFloat(cs.borderLeftWidth) || 0, bt = parseFloat(cs.borderTopWidth) || 0;
+                var pl = parseFloat(cs.paddingLeft) || 0, ptp = parseFloat(cs.paddingTop) || 0;
+                var content = {
+                    x: ib.x + bl + pl, y: ib.y + bt + ptp,
+                    w: el.clientWidth - pl - (parseFloat(cs.paddingRight) || 0),
+                    h: el.clientHeight - ptp - (parseFloat(cs.paddingBottom) || 0)
+                };
+                if (content.w > 0 && content.h > 0) {
+                    add("images", content.y, content.h, { box: content, img: el, viewBox: cs.objectViewBox || el.style.objectViewBox || "" });
+                }
+            }
+
+            // list markers
+            if (el.tagName === "LI" && el.hasAttribute("data-marker")) {
+                var marker = el.getAttribute("data-marker");
+                if (marker) {
+                    var m = markerItem(el, marker, frame);
+                    if (m) add("extras", m.baseline - m.size, m.size * 1.4, m);
+                }
+            }
+
+            // tab leaders
+            if (el.classList.contains("doc-tab") && el.getAttribute("data-leader")) {
+                var lb = frame.box(el.getBoundingClientRect());
+                var leader = el.getAttribute("data-leader");
+                var met = C.fontMetricsOf(cs);
+                // the tab's own line: its baseline is its box's, set by the text
+                add("extras", lb.y, lb.h, {
+                    kind: "leader", box: lb,
+                    ch: leader === "hyphen" ? "-" : (leader === "underscore" ? "_" : "."),
+                    names: C.familyList(cs.fontFamily),
+                    size: parseFloat(cs.fontSize) || 14.67,
+                    bold: (parseInt(cs.fontWeight, 10) || 400) >= 600,
+                    color: parseColor(cs.color) || PDFLib.rgb(0, 0, 0),
+                    baseline: lb.y + (lb.h - (met.ascent + met.descent)) / 2 + met.ascent
+                });
+            }
+
+            // the footnote separator rule
+            if (el.classList.contains("doc-fn-sep")) {
+                var after = window.getComputedStyle(el, "::after");
+                var sb = frame.box(el.getBoundingClientRect());
+                var lw = parseFloat(after.borderTopWidth) || 1;
+                add("extras", sb.y, sb.h, {
+                    kind: "rule",
+                    box: { x: sb.x, y: sb.y + (parseFloat(after.top) || 0), w: parseFloat(after.width) || 192, h: lw },
+                    fill: parseFill(after.borderTopColor) || { c: PDFLib.rgb(0, 0, 0), a: 1 }
+                });
+            }
+        }
+
+        // text, as line fragments at their baselines
+        var origin = { left: frame.left, top: frame.top, x: 0, y: 0 };
+        var runs = C.collectRuns(o.pageEl, origin).filter(function (run) {
+            return run.text.trim() !== "";
+        });
+        runs.forEach(function (run) {
+            var top = run.rect.top - frame.top;
+            var h = run.rect.bottom - run.rect.top;
+            var idx = pageIndexOf(pages, sheetH, top + h / 2);
+            perPage[idx].texts.push(run);
+        });
+        return { perPage: perPage, runs: runs };
+    }
+
+    /* A marker is the li's ::before: drawn in its font, hanging to the left
+       of the item's text and sitting on the item's first baseline. */
+    function markerItem(li, text, frame) {
+        var before = window.getComputedStyle(li, "::before");
+        var size = parseFloat(before.fontSize) || parseFloat(window.getComputedStyle(li).fontSize) || 14.67;
+        var lr = li.getBoundingClientRect();
+        var lcs = window.getComputedStyle(li);
+        var left = lr.left - frame.left + (parseFloat(before.left) || 0);
+        // the first line's baseline: from the first text in the item, else
+        // from the item's own line box
+        var baseline = null;
+        var walker = document.createTreeWalker(li, NodeFilter.SHOW_TEXT, null);
+        var n;
+        while ((n = walker.nextNode())) {
+            if (!n.nodeValue.trim()) continue;
+            if (n.parentElement.closest("ol,ul") !== li.parentElement && n.parentElement.closest("li") !== li) continue;
+            var range = document.createRange();
+            range.selectNodeContents(n);
+            var rects = range.getClientRects();
+            if (!rects.length) continue;
+            var pcs = window.getComputedStyle(n.parentElement);
+            var met = C.fontMetricsOf(pcs);
+            var rh = rects[0].bottom - rects[0].top;
+            baseline = rects[0].top - frame.top + (rh - (met.ascent + met.descent)) / 2 + met.ascent;
+            break;
+        }
+        if (baseline === null) {
+            var lm = C.fontMetricsOf(lcs);
+            var lh = parseFloat(lcs.lineHeight) || size * 1.2;
+            baseline = lr.top - frame.top + (parseFloat(lcs.paddingTop) || 0) + (lh - (lm.ascent + lm.descent)) / 2 + lm.ascent;
+        }
+        return {
+            kind: "marker", text: text, x: left, baseline: baseline, size: size,
+            shape: bulletShape(text, before, size, left, baseline),
+            names: C.familyList(before.fontFamily),
+            bold: (parseInt(before.fontWeight, 10) || 400) >= 600,
+            italic: before.fontStyle === "italic",
+            color: parseColor(before.color) || PDFLib.rgb(0, 0, 0)
+        };
+    }
+
+    /* A geometric bullet (a disc, a ring, a square) is drawn as the shape
+       it is, at the ink box the browser gives the glyph. Written as text it
+       would land in whichever shipped face has the character - for a disc
+       that is a CJK face, whose full-width disc is twice the size of the
+       one Arial draws on screen. */
+    var SHAPES = {
+        0x25CF: "disc", 0x2022: "disc", 0x25CB: "ring", 0x25E6: "ring",
+        0x25A0: "square", 0x25AA: "square", 0x25A1: "box", 0x25AB: "box",
+        0x25C6: "diamond", 0x25C7: "hollowDiamond"
+    };
+    var MEASURE_PX = 200;
+    var shapeCtx = null;
+    function bulletShape(text, cs, size, left, baseline) {
+        var t = String(text || "");
+        if (t.length !== 1 || !SHAPES[t.charCodeAt(0)]) return null;
+        try {
+            if (!shapeCtx) shapeCtx = document.createElement("canvas").getContext("2d");
+            shapeCtx.font = (cs.fontStyle || "normal") + " " + (cs.fontWeight || "400") + " " +
+                MEASURE_PX + "px " + cs.fontFamily;
+            var m = shapeCtx.measureText(t);
+            var k = size / MEASURE_PX;
+            var l = left - m.actualBoundingBoxLeft * k;
+            var r = left + m.actualBoundingBoxRight * k;
+            var top = baseline - m.actualBoundingBoxAscent * k;
+            var bottom = baseline + m.actualBoundingBoxDescent * k;
+            if (!(r > l) || !(bottom > top)) return null;
+            return { kind: SHAPES[t.charCodeAt(0)], x: l, y: top, w: r - l, h: bottom - top };
+        } catch (e) {
+            return null;
+        }
+    }
+
+    /* ---------------- fonts ---------------- */
+
+    /* Which shipped faces the characters need is found out here, in the
+       page, before the display list is written: the list names the font of
+       every piece of text, and a character no face can show is sent as a
+       picture instead. Fetching and parsing a face is what answers "does it
+       have this glyph" (a pass per round of discoveries, as Slides does);
+       embedding it is left to the worker. */
+    function loadCoverage(runs, markers, leaders, fonts) {
+        var MAX_PASSES = OfficeFonts.FALLBACK.length + 2;
+        function each(text, names, bold, italic, fn) {
+            for (var i = 0; i < text.length; i++) {
+                var cp = text.codePointAt(i);
+                if (cp > 0xFFFF) i++;
+                fn(C.resolveChar(cp, names, fonts, bold, italic), bold, italic);
+            }
+        }
+        function eachChar(fn) {
+            runs.forEach(function (r) { each(r.text, r.names, r.weight >= 600, r.italic, fn); });
+            markers.forEach(function (m) { if (!m.shape) each(m.text, m.names, m.bold, m.italic, fn); });
+            leaders.forEach(function (l) { each(l.ch, l.names, l.bold, false, fn); });
+        }
+        function pass(n) {
+            var asked = false;
+            eachChar(function (res, bold, italic) {
+                if (res && res.need) { fonts.want(res.need, bold, italic); asked = true; }
+            });
+            if (!asked || n >= MAX_PASSES) return fonts.ready();
+            return fonts.ready().then(function () { return pass(n + 1); });
+        }
+        return pass(0);
+    }
+
+    function absUrl(u) {
+        try { return new URL(u, document.baseURI).href; } catch (e) { return u; }
+    }
+
+    // the pieces a line fragment is spelled in, as display-list fonts
+    function segsFor(text, names, fonts, bold, italic) {
+        var segs = C.segmentText(text, names, fonts, bold, italic);
+        if (!segs || !segs.length) return null;
+        var out = [];
+        for (var i = 0; i < segs.length; i++) {
+            var res = segs[i].res;
+            if (res.std) {
+                out.push([segs[i].text, "s:" + res.std + ":" + (bold ? 1 : 0) + ":" + (italic ? 1 : 0), false]);
+            } else {
+                var face = OfficeFonts.faceFor(res.shipped, bold, italic);
+                if (!face) return null;
+                out.push([segs[i].text, "f:" + absUrl(face.url), !!face.synthBold]);
+            }
+        }
+        return out;
+    }
+
+    /* ---------------- the display list ---------------- */
+
+    function rgbOf(c) {
+        return c ? [c.red, c.green, c.blue] : [0, 0, 0];
+    }
+
+    // object-view-box: inset(t% r% b% l%) -> the kept fraction of the source
+    function parseInset(v) {
+        var m = /inset\(\s*([-\d.]+)%?\s*([-\d.]+)?%?\s*([-\d.]+)?%?\s*([-\d.]+)?%?\s*\)/.exec(v || "");
+        if (!m) return null;
+        var t = parseFloat(m[1]) || 0;
+        var rr = m[2] !== undefined ? parseFloat(m[2]) : t;
+        var bo = m[3] !== undefined ? parseFloat(m[3]) : t;
+        var l = m[4] !== undefined ? parseFloat(m[4]) : rr;
+        if (!(t || rr || bo || l)) return null;
+        return { t: t / 100, r: rr / 100, b: bo / 100, l: l / 100 };
+    }
+
+    function borderRect(b) {
+        var x = b.box.x, y = b.box.y, w = b.box.w, h = b.box.h, t = b.w;
+        if (b.centred) {
+            switch (b.side) {
+                case "Top": return [x - t / 2, y - t / 2, w + t, t];
+                case "Bottom": return [x - t / 2, y + h - t / 2, w + t, t];
+                case "Left": return [x - t / 2, y - t / 2, t, h + t];
+                default: return [x + w - t / 2, y - t / 2, t, h + t];
+            }
+        }
+        switch (b.side) {
+            case "Top": return [x, y, w, t];
+            case "Bottom": return [x, y + h - t, w, t];
+            case "Left": return [x, y, t, h];
+            default: return [x + w - t, y, t, h];
+        }
+    }
+
+    function shapeOps(s, color, dy) {
+        var col = rgbOf(color);
+        var cx = s.x + s.w / 2, cy = s.y - dy + s.h / 2, top = s.y - dy;
+        // the stroke of a hollow glyph, about what Arial's ring has
+        var stroke = Math.max(0.5, Math.min(s.w, s.h) * 0.1);
+        switch (s.kind) {
+            case "disc": return ["ellipse", cx, cy, s.w / 2, s.h / 2, col, 0];
+            case "ring": return ["ellipse", cx, cy, (s.w - stroke) / 2, (s.h - stroke) / 2, col, stroke];
+            case "square": return ["rect", s.x, top, s.w, s.h, col, 1];
+            case "box": return ["frame", s.x + stroke / 2, top + stroke / 2, s.w - stroke, s.h - stroke, col, stroke];
+            default:
+                return ["poly", [cx, top, s.x + s.w, cy, cx, top + s.h, s.x, cy], col, s.kind === "diamond" ? 0 : stroke];
+        }
+    }
+
+    function canvasBytes(canvas) {
+        return new Promise(function (resolve) {
+            try {
+                canvas.toBlob(function (blob) {
+                    if (!blob) { resolve(null); return; }
+                    blob.arrayBuffer().then(function (b) { resolve(new Uint8Array(b)); }, function () { resolve(null); });
+                }, "image/png");
+            } catch (e) { resolve(null); }
+        });
+    }
+    function sniff(bytes) {
+        if (bytes.length > 3 && bytes[0] === 0xFF && bytes[1] === 0xD8) return "jpg";
+        if (bytes.length > 8 && bytes[0] === 0x89 && bytes[1] === 0x50 && bytes[2] === 0x4E && bytes[3] === 0x47) return "png";
+        return null;
+    }
+    // a picture pdf-lib cannot embed as it is (GIF, WebP, SVG, BMP) goes in
+    // as a PNG of itself at its natural size
+    function rasterPicture(img) {
+        try {
+            var c = document.createElement("canvas");
+            c.width = Math.max(1, img.naturalWidth || img.width);
+            c.height = Math.max(1, img.naturalHeight || img.height);
+            c.getContext("2d").drawImage(img, 0, 0, c.width, c.height);
+            return canvasBytes(c);
+        } catch (e) {
+            return Promise.resolve(null);
+        }
+    }
+    function pictureSpec(img) {
+        var src = img.currentSrc || img.src || "";
+        var m = /^data:([^;,]+)/.exec(src);
+        if (m) {
+            if (/^image\/(jpe?g|png)$/i.test(m[1])) return Promise.resolve({ src: src });
+            return rasterPicture(img).then(function (b) { return b ? { bytes: b } : null; });
+        }
+        return fetch(src).then(function (r) {
+            if (!r.ok) throw new Error("cannot read " + src);
+            return r.arrayBuffer();
+        }).then(function (buf) {
+            var bytes = new Uint8Array(buf);
+            if (sniff(bytes)) return { bytes: bytes };
+            return rasterPicture(img).then(function (b) { return b ? { bytes: b } : null; });
+        }).catch(function () {
+            return rasterPicture(img).then(function (b) { return b ? { bytes: b } : null; });
+        });
+    }
+
+    /* a run no font here can show (emoji, a script not shipped) is drawn
+       as a picture of its own text, by the browser's own text renderer */
+    function rasterRun(run) {
+        var w = run.rect.right - run.rect.left, h = run.rect.bottom - run.rect.top;
+        if (w <= 0 || h <= 0) return Promise.resolve(null);
+        var k = C.RASTER_SCALE;
+        var c = document.createElement("canvas");
+        c.width = Math.ceil(w * k);
+        c.height = Math.ceil(h * k);
+        var g = c.getContext("2d");
+        g.scale(k, k);
+        g.font = (run.italic ? "italic " : "") + run.weight + " " + run.size + "px " + run.names.map(function (n) {
+            return /[\s"']/.test(n) ? '"' + n.replace(/"/g, "") + '"' : n;
+        }).join(",");
+        g.fillStyle = run.color;
+        g.textBaseline = "alphabetic";
+        var glyphH = run.metrics.ascent + run.metrics.descent;
+        g.fillText(run.text, 0, (h - glyphH) / 2 + run.metrics.ascent);
+        return canvasBytes(c);
+    }
+
+    function breathe() {
+        return new Promise(function (res) { setTimeout(res, 0); });
+    }
+
+    /* displayList turns what collect() measured into the job pdfdraw.js
+       draws: every coordinate relative to its own sheet, every colour a
+       triple, every font a reference, every picture an entry of its own. */
+    function displayList(o, data, fonts, frame) {
+        var job = { title: o.title || "", sheetW: o.sheetW, sheetH: o.sheetH, images: {}, pages: [] };
+        var imageIds = new Map();
+        var pending = [];
+        var nextId = 0;
+        function pictureId(img) {
+            var key = img.currentSrc || img.src || img;
+            if (imageIds.has(key)) return imageIds.get(key);
+            var id = "i" + (nextId++);
+            imageIds.set(key, id);
+            pending.push(pictureSpec(img).then(function (spec) { if (spec) job.images[id] = spec; }));
+            return id;
+        }
+        var chain = Promise.resolve();
+        o.pages.forEach(function (page, i) {
+            chain = chain.then(function () {
+                var items = data.perPage[i];
+                var dy = page.sheetTop;
+                var ops = [];
+                job.pages.push(ops);
+                items.fills.forEach(function (f) {
+                    ops.push(["rect", f.box.x, f.box.y - dy, f.box.w, f.box.h, rgbOf(f.fill.c), f.fill.a]);
+                });
+                items.images.forEach(function (im) {
+                    ops.push(["image", pictureId(im.img), im.box.x, im.box.y - dy, im.box.w, im.box.h, parseInset(im.viewBox)]);
+                });
+                items.borders.forEach(function (b) {
+                    var r = borderRect(b);
+                    ops.push(["rect", r[0], r[1] - dy, r[2], r[3], rgbOf(b.fill.c), b.fill.a]);
+                });
+                var rasters = [];
+                items.texts.forEach(function (r) {
+                    var x = r.origin.x + (r.rect.left - r.origin.left);
+                    var top = r.origin.y + (r.rect.top - r.origin.top) - dy;
+                    var lineH = r.rect.bottom - r.rect.top;
+                    var domW = r.rect.right - r.rect.left;
+                    var bg = C.parseFill(r.background);
+                    if (bg) ops.push(["rect", x, top, domW, lineH, rgbOf(bg.c), bg.a]);
+                    var segs = segsFor(r.text, r.names, fonts, r.weight >= 600, r.italic);
+                    if (!segs) {
+                        rasters.push({ run: r, at: ops.length, x: x, top: top, w: domW, h: lineH });
+                        ops.push(null);
+                        return;
+                    }
+                    var glyphH = r.metrics.ascent + r.metrics.descent;
+                    var baseline = (lineH - glyphH) / 2 + r.metrics.ascent;
+                    var deco = null;
+                    if (r.underline || r.strike) {
+                        var yOff = r.underline ? baseline + r.size * 0.11 : baseline - r.size * 0.28;
+                        deco = [top + yOff, Math.max(0.7, r.size * 0.06), domW];
+                    }
+                    // a fragment that starts or ends on a space cannot be fitted
+                    // to its rect: the browser collapses those, the measure does not
+                    ops.push(["text", x, top + baseline, r.size, rgbOf(C.parseColor(r.color)),
+                        /^\s|\s$/.test(r.text) ? 0 : domW, segs, deco]);
+                });
+                items.extras.forEach(function (x) {
+                    if (x.kind === "marker") {
+                        if (x.shape) {
+                            ops.push(shapeOps(x.shape, x.color, dy));
+                        } else {
+                            var ms = segsFor(x.text, x.names, fonts, x.bold, x.italic);
+                            if (ms) ops.push(["text", x.x, x.baseline - dy, x.size, rgbOf(x.color), 0, ms, null]);
+                        }
+                    } else if (x.kind === "leader") {
+                        var ls = segsFor(x.ch, x.names, fonts, x.bold, false);
+                        if (ls) ops.push(["leader", x.box.x, x.box.w, x.baseline - dy, x.size, rgbOf(x.color), x.ch, ls[0][1]]);
+                    } else if (x.kind === "rule") {
+                        ops.push(["rect", x.box.x, x.box.y - dy, x.box.w, x.box.h, rgbOf(x.fill.c), x.fill.a]);
+                    }
+                });
+                // text no font can show becomes a picture of itself, in place
+                var rs = rasters.map(function (it) {
+                    return rasterRun(it.run).then(function (bytes) {
+                        if (!bytes) return;
+                        var id = "r" + (nextId++);
+                        job.images[id] = { bytes: bytes };
+                        ops[it.at] = ["image", id, it.x, it.top, it.w, it.h, null];
+                    });
+                });
+                return Promise.all(rs).then(function () {
+                    for (var k = ops.length - 1; k >= 0; k--) if (!ops[k]) ops.splice(k, 1);
+                    return breathe();
+                });
+            });
+        });
+        return chain.then(function () {
+            return Promise.all(pending);
+        }).then(function () {
+            return job;
+        });
+    }
+
+    /* ---------------- running the job ---------------- */
+
+    var WORKER_URL = "../common/pdfworker.js";
+
+    /* The worker is where the time goes (pictures, fonts, compression).
+       When one cannot be started at all - an old browser, a page opened
+       from disk - the same code runs here instead, just less politely. */
+    function renderJob(job, onProgress) {
+        return new Promise(function (resolve, reject) {
+            var worker;
+            var started = false;
+            function inPage() {
+                C.loadFontkit().then(function (fk) {
+                    return OfficePdfDraw.render(job, { fontkit: fk, onProgress: onProgress });
+                }).then(resolve, reject);
+            }
+            try {
+                worker = new Worker(WORKER_URL);
+            } catch (e) {
+                inPage();
+                return;
+            }
+            worker.onmessage = function (e) {
+                var m = e.data || {};
+                started = true;
+                if (m.type === "progress") {
+                    if (onProgress) onProgress(m.done, m.total, m.stage);
+                } else if (m.type === "done") {
+                    worker.terminate();
+                    resolve(m.bytes);
+                } else if (m.type === "error") {
+                    worker.terminate();
+                    reject(new Error(m.message || "the PDF could not be written"));
+                }
+            };
+            worker.onerror = function (e) {
+                if (e && e.preventDefault) e.preventDefault();
+                worker.terminate();
+                if (!started && typeof OfficePdfDraw !== "undefined") inPage();
+                else reject(new Error((e && e.message) || "the PDF worker failed"));
+            };
+            var transfer = [];
+            Object.keys(job.images).forEach(function (id) {
+                var b = job.images[id].bytes;
+                if (b && b.buffer && transfer.indexOf(b.buffer) < 0) transfer.push(b.buffer);
+            });
+            worker.postMessage({ job: job }, transfer);
+        });
+    }
+
+    function waitForImages(root) {
+        var imgs = Array.prototype.slice.call(root.querySelectorAll("img"));
+        var pending = imgs.filter(function (im) { return !im.complete; });
+        var fontsP = OfficeFonts.preload().then(function () {
+            return document.fonts && document.fonts.ready ? document.fonts.ready : null;
+        });
+        return Promise.all([fontsP].concat(pending.map(function (im) {
+            return new Promise(function (res) {
+                im.addEventListener("load", function () { res(); });
+                im.addEventListener("error", function () { res(); });
+                setTimeout(res, 8000);
+            });
+        })));
+    }
+
+    /* build runs an export in three steps, and only the first holds the
+       page, for as long as it takes to read the layout:
+
+         1. snapshot  o.enter() puts the page in its export state, and what
+                      is on each sheet is measured - boxes, text runs,
+                      markers - then o.leave() gives the page back. From
+                      here on the document may be edited: nothing below
+                      looks at it again.
+         2. prepare   fonts are checked for every character and the display
+                      list is written (in steps, between which the page runs)
+         3. render    a worker turns the list into the file
+
+       o: { pageEl, pages() | pages, sheetW, sheetH, title,
+            enter(), leave(), onProgress(done, total, stage) }
+       stage is "measure", "prepare", "page" or "save". */
+    function build(o) {
+        if (typeof PDFLib === "undefined") return Promise.reject(new Error("the PDF library failed to load"));
+        if (!o || !o.pageEl) return Promise.reject(new Error("nothing to export"));
+        var progress = o.onProgress || function () { };
+        var data, snap, fonts;
+        progress(0, 1, "measure");
+        return waitForImages(o.pageEl).then(function () {
+            return C.loadFontkit();
+        }).then(function (fontkit) {
+            try {
+                if (o.enter) o.enter();
+                snap = {
+                    pageEl: o.pageEl,
+                    pages: (typeof o.pages === "function" ? o.pages() : o.pages).slice(),
+                    sheetW: o.sheetW, sheetH: o.sheetH, title: o.title
+                };
+                if (!snap.pages.length) throw new Error("nothing to export");
+                data = collect(snap, new Frame(o.pageEl));
+            } finally {
+                if (o.leave) o.leave();
+            }
+            fonts = C.makeFonts(null, fontkit);
+            var markers = [], leaders = [];
+            data.perPage.forEach(function (p) {
+                p.extras.forEach(function (x) {
+                    if (x.kind === "marker") markers.push(x);
+                    else if (x.kind === "leader") leaders.push(x);
+                });
+            });
+            progress(0, 1, "prepare");
+            return loadCoverage(data.runs, markers, leaders, fonts);
+        }).then(function () {
+            return displayList(snap, data, fonts);
+        }).then(function (job) {
+            data = null;
+            return renderJob(job, progress);
+        });
+    }
+
+    return { build: build };
+})();

+ 5 - 0
src/web/Office/docs/index.html

@@ -69,6 +69,11 @@
     <!-- Hidden input for "Insert image from this device" -->
     <!-- Hidden input for "Insert image from this device" -->
     <input type="file" id="deviceImageInput" accept="image/*" style="display:none;">
     <input type="file" id="deviceImageInput" accept="image/*" style="display:none;">
 
 
+    <script src="../common/lib/pdf-lib.min.js"></script>
+    <script src="../common/pdfcore.js"></script>
+    <script src="../common/pdfdraw.js"></script>
+    <script src="docs_layout.js"></script>
+    <script src="docs_pdf.js"></script>
     <script src="docs.js"></script>
     <script src="docs.js"></script>
 </body>
 </body>
 </html>
 </html>

+ 1 - 0
src/web/Office/slides/index.html

@@ -54,6 +54,7 @@
     <input type="file" id="slDeviceImage" accept="image/*" multiple style="display:none;">
     <input type="file" id="slDeviceImage" accept="image/*" multiple style="display:none;">
     <script src="slides_shapes.js"></script>
     <script src="slides_shapes.js"></script>
     <script src="slides_image.js"></script>
     <script src="slides_image.js"></script>
+    <script src="../common/pdfcore.js"></script>
     <script src="slides_pdf.js"></script>
     <script src="slides_pdf.js"></script>
     <script src="slides.js"></script>
     <script src="slides.js"></script>
     <script src="present.js"></script>
     <script src="present.js"></script>

+ 14 - 745
src/web/Office/slides/slides_pdf.js

@@ -60,313 +60,22 @@ var SlidesPdf = (function () {
 
 
     // the slide is 960x540 css px; a PDF point is 1/72", a css px 1/96",
     // the slide is 960x540 css px; a PDF point is 1/72", a css px 1/96",
     // so the page is 720x405 pt - the 10" x 5.625" of the pptx slide size
     // so the page is 720x405 pt - the 10" x 5.625" of the pptx slide size
-    var PX_TO_PT = 0.75;
     var SLIDE_W = 960, SLIDE_H = 540;
     var SLIDE_W = 960, SLIDE_H = 540;
-    var RASTER_SCALE = 3;      // device pixels per css px for a fallback raster
 
 
-    /* ---------------- fonts ---------------- */
-
-    // where fontkit lives; it is fetched only when an export runs
-    var FONTKIT_URL = "../common/lib/fontkit.umd.min.js";
-
-    /* Families that are metric-compatible with a standard PDF font, so
-       text set in them lands in exactly the same place as on screen. */
-    var STD_FAMILIES = {
-        "helvetica": "Helvetica", "arial": "Helvetica", "liberation sans": "Helvetica",
-        "arimo": "Helvetica", "sans-serif": "Helvetica", "nimbus sans": "Helvetica",
-        "times": "TimesRoman", "times new roman": "TimesRoman", "serif": "TimesRoman",
-        "liberation serif": "TimesRoman", "tinos": "TimesRoman", "nimbus roman": "TimesRoman",
-        "courier": "Courier", "courier new": "Courier", "monospace": "Courier",
-        "liberation mono": "Courier", "cousine": "Courier", "nimbus mono": "Courier"
-    };
-    var STD_VARIANTS = {
-        Helvetica: ["Helvetica", "HelveticaBold", "HelveticaOblique", "HelveticaBoldOblique"],
-        TimesRoman: ["TimesRoman", "TimesRomanBold", "TimesRomanItalic", "TimesRomanBoldItalic"],
-        Courier: ["Courier", "CourierBold", "CourierOblique", "CourierBoldOblique"]
-    };
-    var GENERICS = { "sans-serif": 1, "serif": 1, "monospace": 1, "cursive": 1, "fantasy": 1 };
-
-    /* WinAnsi is what the standard fonts can encode. A handful of code
-       points above U+00FF are in it too, but keeping to the Latin-1 range
-       is the rule that can be checked without a table. */
-    function winAnsiCp(cp) {
-        return cp >= 32 && cp <= 255;
-    }
-    /* haveFamily reports whether a family is actually installed. It has to
-       be measured: document.fonts.check() answers "is it loaded", and for a
-       local family Chrome says yes whatever name you give it. The reliable
-       test is the old one - render a probe string backed by two different
-       generics; a family that exists overrides both and comes out a
-       different width from each, one that does not falls through to the
-       generic and matches it exactly.
-
-       This matters because a deck may ask for a font the machine does not
-       have: the browser laid the text out in the fallback, so the fallback
-       is what the PDF must match - and that may well be one it can show. */
-    var PROBE = "mmmmmmmmwwwwwwwwiiiiiiiil1I0Oo";
-    var familyKnown = {};
-    var probeCtx = null;
-    var probeBase = null;
-    function haveFamily(name) {
-        if (familyKnown[name] !== undefined) return familyKnown[name];
-        try {
-            if (!probeCtx) {
-                probeCtx = document.createElement("canvas").getContext("2d");
-                probeBase = {};
-                ["monospace", "serif"].forEach(function (g) {
-                    probeCtx.font = '72px ' + g;
-                    probeBase[g] = probeCtx.measureText(PROBE).width;
-                });
-            }
-            var found = true;
-            ["monospace", "serif"].forEach(function (g) {
-                probeCtx.font = '72px "' + name.replace(/"/g, "") + '", ' + g;
-                if (probeCtx.measureText(PROBE).width === probeBase[g]) found = false;
-            });
-            familyKnown[name] = found;
-        } catch (e) {
-            familyKnown[name] = true;
-        }
-        return familyKnown[name];
-    }
-
-    /* familyList turns a computed font-family into the list the browser
-       walks, with the shipped faces on the end. The tail matters for old
-       content: a <font face="..."> names one family and nothing else, and
-       without it a character that family has no glyph for would have
-       nowhere to go. */
-    function familyList(cssFamily) {
-        var out = [], seen = {};
-        function add(n) {
-            n = String(n).trim().replace(/^["']|["']$/g, "");
-            if (!n || seen[n.toLowerCase()]) return;
-            seen[n.toLowerCase()] = true;
-            out.push(n);
-        }
-        String(cssFamily || "").split(",").forEach(add);
-        OfficeFonts.FALLBACK.forEach(add);
-        return out;
-    }
-
-    /* fontkit is what lets pdf-lib embed a font file of our own. It is the
-       largest script the app has and only an export needs it, so it is
-       fetched on the first export and not before. */
-    var fontkitPromise = null;
-    function loadFontkit() {
-        if (fontkitPromise) return fontkitPromise;
-        if (window.fontkit) {
-            fontkitPromise = Promise.resolve(window.fontkit);
-            return fontkitPromise;
-        }
-        fontkitPromise = new Promise(function (resolve, reject) {
-            var el = document.createElement("script");
-            el.src = FONTKIT_URL;
-            el.onload = function () {
-                if (window.fontkit) resolve(window.fontkit);
-                else reject(new Error("the font toolkit did not load"));
-            };
-            el.onerror = function () { reject(new Error("the font toolkit did not load")); };
-            document.head.appendChild(el);
-        });
-        return fontkitPromise;
-    }
-
-    /* makeFonts is the document's font supply.
-
-       A standard font is there for the asking. A shipped face goes through
-       two stages, and the split matters:
-
-         want()  fetches the file and parses it, which is what answers "does
-                 this face have a glyph for this character". Asynchronous,
-                 so a slide says what it needs, waits (ready), then draws.
-         use()   puts it in the PDF. Only faces that really get drawn with
-                 may be embedded: the embedder subsets a font down to the
-                 glyphs that were asked of it, and a subset of nothing is
-                 not a font any more - a CFF one fails outright on save.
-
-       Drawing itself stays synchronous, which is what lets a fragment be
-       measured and placed in one pass. */
-    function makeFonts(pdfDoc, fontkit) {
-        var std = {};
-        var faces = {};        // url -> { kit, font? }, or null when it failed
-        var asked = {};        // url -> Promise, set the moment it is wanted
-        var wanted = [];
-
-        function stdFont(family, bold, italic) {
-            var names = STD_VARIANTS[family] || STD_VARIANTS.Helvetica;
-            var key = names[(bold ? 1 : 0) + (italic ? 2 : 0)];
-            if (!std[key]) std[key] = pdfDoc.embedStandardFont(PDFLib.StandardFonts[key]);
-            return std[key];
-        }
-
-        function want(family, bold, italic) {
-            var face = OfficeFonts.faceFor(family, bold, italic);
-            if (!face || asked[face.url]) return;
-            asked[face.url] = fetch(face.url).then(function (r) {
-                if (!r.ok) throw new Error("cannot read " + face.url);
-                return r.arrayBuffer();
-            }).then(function (buf) {
-                var bytes = new Uint8Array(buf);
-                faces[face.url] = { bytes: bytes, kit: fontkit.create(bytes) };
-            }, function () {
-                // a font that will not load is not a reason to fail the
-                // export: the next family in the stack gets the character
-                faces[face.url] = null;
-            });
-            wanted.push(asked[face.url]);
-        }
-
-        function use(family, bold, italic) {
-            var face = OfficeFonts.faceFor(family, bold, italic);
-            if (!face) return;
-            var rec = faces[face.url];
-            if (!rec || rec.font || rec.embedding) return;
-            rec.embedding = pdfDoc.embedFont(rec.bytes, { subset: true })
-                .then(function (font) { rec.font = font; });
-            wanted.push(rec.embedding);
-        }
-
-        function ready() {
-            var all = wanted;
-            wanted = [];
-            if (!all.length) return Promise.resolve();
-            return Promise.all(all).then(function () { });
-        }
-
-        // shipped hands back a loaded face, null while it is not there
-        function shipped(family, bold, italic) {
-            var face = OfficeFonts.faceFor(family, bold, italic);
-            if (!face) return null;
-            var rec = faces[face.url];
-            if (!rec) return null;
-            return {
-                font: rec.font, kit: rec.kit,
-                synthBold: face.synthBold, synthItalic: face.synthItalic
-            };
-        }
-
-        // tried says whether asking again could still change the answer
-        function tried(family, bold, italic) {
-            var face = OfficeFonts.faceFor(family, bold, italic);
-            return !face || !!asked[face.url];
-        }
-
-        return {
-            std: stdFont, want: want, use: use,
-            ready: ready, shipped: shipped, tried: tried
-        };
-    }
-
-    /* resolveChar walks a font stack the way the browser does and says what
-       the PDF can put this one character in:
-
-         { std }      one of the 14 standard fonts
-         { shipped }  a face the suite ships, already embedded
-         { need }     a shipped face that is named but not loaded yet, so
-                      the answer is not known until it is
-         null         nothing here can show this character
-
-       A family that is neither - a system font - is stepped over rather
-       than used: its bytes are unreadable, so the character goes to the
-       next entry, which is the shipped face for its script. */
-    function resolveChar(cp, names, fonts, bold, italic) {
-        for (var i = 0; i < names.length; i++) {
-            var name = names[i];
-            var key = name.toLowerCase();
-            if (OfficeFonts.isShipped(name)) {
-                var rec = fonts.shipped(name, bold, italic);
-                if (!rec) {
-                    if (!fonts.tried(name, bold, italic)) return { need: name };
-                    continue;
-                }
-                if (rec.kit && rec.kit.hasGlyphForCodePoint &&
-                    !rec.kit.hasGlyphForCodePoint(cp)) continue;
-                return { shipped: name };
-            }
-            if (STD_FAMILIES[key] && (GENERICS[key] || haveFamily(name)) && winAnsiCp(cp)) {
-                return { std: STD_FAMILIES[key] };
-            }
-        }
-        return null;
-    }
-
-    /* segmentText cuts a fragment into the pieces that share one font, the
-       way a browser does per character. Returns null when any character has
-       nowhere to go, which is the signal to rasterize instead. */
-    function segmentText(text, names, fonts, bold, italic) {
-        var segs = [], cur = null;
-        for (var i = 0; i < text.length; i++) {
-            var cp = text.codePointAt(i);
-            var ch = String.fromCodePoint(cp);
-            if (ch.length > 1) i++;          // a surrogate pair
-            var res = resolveChar(cp, names, fonts, bold, italic);
-            if (!res || res.need) return null;
-            var key = res.std ? "s:" + res.std : "f:" + res.shipped;
-            if (cur && cur.key === key) cur.text += ch;
-            else { cur = { key: key, res: res, text: ch }; segs.push(cur); }
-        }
-        return segs;
-    }
-
-    function faceOf(res, fonts, bold, italic) {
-        if (res.std) return { font: fonts.std(res.std, bold, italic), synthBold: false };
-        var rec = fonts.shipped(res.shipped, bold, italic);
-        return rec ? { font: rec.font, synthBold: rec.synthBold } : null;
-    }
-
-    /* Where the baseline sits is the browser's decision, and the exporter
-       has to ask rather than compute: a system font's metrics are not
-       readable from the page, and even for a font that is, the numbers the
-       file states are not always the ones the browser uses.
-
-       A canvas answers it. measureText reports the ascent and descent the
-       browser resolved for a font stack, which is exactly what it used to
-       lay the text out - so the two cannot drift apart.
-
-       This is also why the run's own rect is the reference: the rects a
-       Range hands back for text are the content box, ascent plus descent
-       tall, not the line box. The baseline is therefore an ascent below the
-       top of the rect, with the halving below for the case where a browser
-       hands back the taller box instead. */
-    var metricsCache = {};
-    var metricsCtx = null;
-    function fontMetricsOf(cs) {
-        var font = cs.fontStyle + " " + cs.fontWeight + " " + cs.fontSize + " " + cs.fontFamily;
-        if (metricsCache[font]) return metricsCache[font];
-        var size = parseFloat(cs.fontSize) || 12;
-        var m = null;
-        try {
-            if (!metricsCtx) metricsCtx = document.createElement("canvas").getContext("2d");
-            metricsCtx.font = font;
-            var tm = metricsCtx.measureText("Hxg");
-            if (tm && tm.fontBoundingBoxAscent !== undefined) {
-                m = { ascent: tm.fontBoundingBoxAscent, descent: tm.fontBoundingBoxDescent };
-            }
-        } catch (e) { /* fall through to the estimate */ }
-        // a browser without the font bounding box: the usual proportions
-        if (!m) m = { ascent: size * 0.9, descent: size * 0.22 };
-        metricsCache[font] = m;
-        return m;
-    }
-
-    /* eachTextNode is the walk both the font pre-pass and the run collector
-       make, kept in one place so they cannot disagree about what counts as
-       text on the slide. */
-    function eachTextNode(rootEl, fn) {
-        var walker = document.createTreeWalker(rootEl, NodeFilter.SHOW_TEXT, null);
-        var node;
-        while ((node = walker.nextNode())) {
-            var text = node.nodeValue;
-            if (!text || !text.trim()) continue;
-            var parent = node.parentElement;
-            if (!parent) continue;
-            var cs = window.getComputedStyle(parent);
-            if (cs.visibility === "hidden" || cs.display === "none") continue;
-            // source newlines and tabs are whitespace the browser already
-            // collapsed; they must not reach a font, but the string has to
-            // keep its length - the line split indexes back into the node
-            fn(node, text.replace(/[\u0000-\u001F\u007F]/g, " "), cs);
-        }
+    /* the shared exporter core (common/pdfcore.js) */
+    var C = OfficePdfCore;
+    var PX_TO_PT = C.PX_TO_PT;
+    var familyList = C.familyList, loadFontkit = C.loadFontkit, makeFonts = C.makeFonts;
+    var resolveChar = C.resolveChar, segmentText = C.segmentText, faceOf = C.faceOf;
+    var eachTextNode = C.eachTextNode, px = C.px, clamp = C.clamp;
+    var parseFill = C.parseFill, parseColor = C.parseColor, contrastOf = C.contrastOf;
+    var collectRuns = C.collectRuns, canDrawAsText = C.canDrawAsText;
+    var drawFragment = C.drawFragment, drawRuns = C.drawRuns;
+    var rasterizeElement = C.rasterizeElement, rasterizeFallback = C.rasterizeFallback;
+    var filteredImageData = C.filteredImageData, makeImageEmbedder = C.makeImageEmbedder;
+    // a slide page: the core's drawing context at the slide's height
+    function Page(page, pdfDoc, fonts) {
+        return new C.Page(page, pdfDoc, fonts, SLIDE_H);
     }
     }
 
 
     /* prepareFonts loads what this slide is about to need, and then checks
     /* prepareFonts loads what this slide is about to need, and then checks
@@ -424,114 +133,6 @@ var SlidesPdf = (function () {
         return pass(0).then(embedUsed);
         return pass(0).then(embedUsed);
     }
     }
 
 
-    /* ---------------- small helpers ---------------- */
-
-    function px(v) { return v * PX_TO_PT; }
-    function clamp(v, a, b) { return Math.max(a, Math.min(b, v)); }
-
-    /* parseFill returns both halves of a CSS colour: the colour itself and
-       its alpha. The alpha matters - a table's header band is a translucent
-       wash of the theme accent, and dropping it turns a tint into a slab. */
-    function parseFill(css) {
-        if (!css) return null;
-        var m = /^rgba?\(([^)]+)\)$/i.exec(String(css).trim());
-        if (m) {
-            var p = m[1].split(",").map(function (x) { return parseFloat(x); });
-            var a = p.length >= 4 ? clamp(p[3], 0, 1) : 1;
-            if (a === 0) return null;
-            return {
-                c: PDFLib.rgb(clamp(p[0] / 255, 0, 1), clamp(p[1] / 255, 0, 1), clamp(p[2] / 255, 0, 1)),
-                a: a
-            };
-        }
-        var t = String(css).trim();
-        var h = /^#([0-9a-f]{3}|[0-9a-f]{6}|[0-9a-f]{8})$/i.exec(t);
-        if (!h) return null;
-        var v = h[1];
-        var alpha = 1;
-        if (v.length === 8) { alpha = parseInt(v.substring(6, 8), 16) / 255; v = v.substring(0, 6); }
-        if (v.length === 3) v = v[0] + v[0] + v[1] + v[1] + v[2] + v[2];
-        if (alpha === 0) return null;
-        var n = parseInt(v, 16);
-        return {
-            c: PDFLib.rgb(((n >> 16) & 255) / 255, ((n >> 8) & 255) / 255, (n & 255) / 255),
-            a: alpha
-        };
-    }
-    // parseColor is parseFill when only the colour is wanted
-    /* contrastOf answers "what colour shows up on this fill" - only used
-       for a shape's markings when it has no stroke colour of its own */
-    function contrastOf(css) {
-        var f = parseFill(css);
-        if (!f) return "#333333";
-        var lum = 0.299 * f.c.red + 0.587 * f.c.green + 0.114 * f.c.blue;
-        return lum > 0.6 ? "#333333" : "#ffffff";
-    }
-
-    function parseColor(css) {
-        var f = parseFill(css);
-        return f ? f.c : null;
-    }
-
-    function dataUrlBytes(src) {
-        var comma = String(src || "").indexOf(",");
-        if (comma < 0) return null;
-        var head = src.substring(0, comma);
-        if (head.indexOf(";base64") < 0) return null;
-        var bin = atob(src.substring(comma + 1));
-        var out = new Uint8Array(bin.length);
-        for (var i = 0; i < bin.length; i++) out[i] = bin.charCodeAt(i);
-        return { bytes: out, mime: (/^data:([^;]+)/.exec(head) || [])[1] || "" };
-    }
-
-    /* ---------------- the drawing context ---------------- */
-
-    /* Page is a thin wrapper that flips the y axis once: the editor's
-       coordinates run down from the top-left of the slide, a PDF page's run
-       up from the bottom-left, and mixing the two up is the single easiest
-       way to get an export subtly wrong. */
-    function Page(page, pdfDoc, fonts) {
-        this.p = page;
-        this.doc = pdfDoc;
-        this.fonts = fonts;
-        this.gsCache = {};
-    }
-    Page.prototype.y = function (topPx) { return px(SLIDE_H - topPx); };
-
-    Page.prototype.rect = function (x, y, w, h, opts) {
-        this.p.drawRectangle({
-            x: px(x), y: this.y(y + h), width: px(w), height: px(h),
-            color: opts.fill || undefined,
-            borderColor: opts.stroke || undefined,
-            borderWidth: opts.strokeW ? px(opts.strokeW) : undefined,
-            borderDashArray: opts.dash ? [px(opts.strokeW * 3), px(opts.strokeW * 2)] : undefined,
-            opacity: opts.fillOpacity !== undefined ? opts.fillOpacity : opts.opacity,
-            borderOpacity: opts.strokeOpacity !== undefined ? opts.strokeOpacity : opts.opacity
-        });
-    };
-
-    // ops pushes raw content-stream operators, which is how the clip paths
-    // and the matrices below are expressed
-    Page.prototype.ops = function (list) {
-        this.p.pushOperators.apply(this.p, list);
-    };
-    Page.prototype.save = function () { this.ops([PDFLib.pushGraphicsState()]); };
-    Page.prototype.restore = function () { this.ops([PDFLib.popGraphicsState()]); };
-
-    // alpha returns the name of an ExtGState for a given opacity, making one
-    // only the first time each distinct value is used on this page
-    Page.prototype.alpha = function (a) {
-        var key = "a" + Math.round(a * 1000);
-        if (!this.gsCache[key]) {
-            var ref = this.doc.context.register(this.doc.context.obj({
-                Type: "ExtGState", ca: a, CA: a
-            }));
-            this.p.node.setExtGState(PDFLib.PDFName.of(key), ref);
-            this.gsCache[key] = true;
-        }
-        return key;
-    };
-
     /* ---------------- geometry -> PDF path operators ---------------- */
     /* ---------------- geometry -> PDF path operators ---------------- */
 
 
     // shapePathOps turns one of the editor's shape outlines into path
     // shapePathOps turns one of the editor's shape outlines into path
@@ -596,275 +197,6 @@ var SlidesPdf = (function () {
         return ops;
         return ops;
     }
     }
 
 
-    /* ---------------- text ---------------- */
-
-    /* A run is one uniformly formatted fragment of a single line, measured
-       off the live DOM: its box, its baseline and the style in force. */
-    function collectRuns(rootEl, origin) {
-        var runs = [];
-        eachTextNode(rootEl, function (node, text, cs) {
-            // one entry per line box the fragment occupies
-            var range = document.createRange();
-            range.selectNodeContents(node);
-            var rects = Array.prototype.slice.call(range.getClientRects());
-            if (!rects.length) return;
-            var shared = {
-                origin: origin,
-                names: familyList(cs.fontFamily),
-                metrics: fontMetricsOf(cs),
-                size: parseFloat(cs.fontSize) || 12,
-                weight: parseInt(cs.fontWeight, 10) || (cs.fontWeight === "bold" ? 700 : 400),
-                italic: cs.fontStyle === "italic" || cs.fontStyle === "oblique",
-                underline: cs.textDecorationLine.indexOf("underline") >= 0,
-                strike: cs.textDecorationLine.indexOf("line-through") >= 0,
-                color: cs.color,
-                background: cs.backgroundColor
-            };
-            splitByLine(node, text, rects).forEach(function (ln) {
-                var r = Object.create(shared);
-                r.text = ln.text;
-                r.rect = ln.rect;
-                runs.push(r);
-            });
-        });
-        return runs;
-    }
-
-    /* splitByLine maps a text node's client rects back onto the substrings
-       that produced them, so each line can be drawn at its own baseline.
-       Character-by-character is the only reliable way: the browser decides
-       where the break went, and only it knows. */
-    function splitByLine(node, text, lineRects) {
-        if (lineRects.length === 1) {
-            return [{ text: text, rect: lineRects[0] }];
-        }
-        var range = document.createRange();
-        var out = [];
-        var cur = "";
-        var curTop = null;
-        var curRect = null;
-        for (var i = 0; i < text.length; i++) {
-            range.setStart(node, i);
-            range.setEnd(node, i + 1);
-            var r = range.getBoundingClientRect();
-            if (r.width === 0 && r.height === 0) { cur += text[i]; continue; }
-            var top = Math.round(r.top * 10) / 10;
-            if (curTop === null || Math.abs(top - curTop) < 0.6) {
-                if (curTop === null) { curTop = top; curRect = { left: r.left, top: r.top, bottom: r.bottom, right: r.right }; }
-                else { curRect.right = Math.max(curRect.right, r.right); }
-                cur += text[i];
-            } else {
-                out.push({ text: cur, rect: curRect });
-                cur = text[i];
-                curTop = top;
-                curRect = { left: r.left, top: r.top, bottom: r.bottom, right: r.right };
-            }
-        }
-        if (cur !== "" && curRect) out.push({ text: cur, rect: curRect });
-        return out.length ? out : [{ text: text, rect: lineRects[0] }];
-    }
-
-    // canDrawAsText is the whole fallback decision, in one place: every
-    // character of every run has to have a font that can show it
-    function canDrawAsText(runs, fonts) {
-        for (var i = 0; i < runs.length; i++) {
-            var r = runs[i];
-            if (!segmentText(r.text, r.names, fonts, r.weight >= 600, r.italic)) return false;
-        }
-        return true;
-    }
-
-    /* drawFragment puts one line fragment on the page as real text, in as
-       many pieces as it takes fonts to spell it.
-
-       fitPx, when it is given, is the width the browser gave the fragment.
-       The text is squeezed or stretched to exactly that with Tz, which
-       costs nothing when the PDF font is the one the browser used (the
-       ratio is 1) and is what keeps a substituted face - a system font we
-       could not embed - from pushing the rest of the line out of place.
-
-       Bold that a shipped face does not have is stroked rather than filled,
-       at the width the browser smears it by. Neither side moves the advance
-       widths, so the two stay in step. */
-    function drawFragment(pg, spec) {
-        var fonts = pg.fonts;
-        var segs = segmentText(spec.text, spec.names, fonts, spec.bold, spec.italic);
-        if (!segs || !segs.length) return 0;
-
-        var total = 0;
-        for (var i = 0; i < segs.length; i++) {
-            var face = faceOf(segs[i].res, fonts, spec.bold, spec.italic);
-            if (!face) return 0;
-            segs[i].face = face;
-            segs[i].w = face.font.widthOfTextAtSize(segs[i].text, spec.sizePx);
-            total += segs[i].w;
-        }
-
-        var scale = 1;
-        if (spec.fitPx > 0 && total > 0) {
-            var ratio = spec.fitPx / total;
-            // a ratio far from 1 means the measurement, not the font, is
-            // wrong (a collapsed space, a transform) - leave it alone
-            if (ratio > 0.5 && ratio < 2 && Math.abs(ratio - 1) > 0.005) scale = ratio;
-        }
-
-        var col = spec.color || PDFLib.rgb(0, 0, 0);
-        var cursor = spec.xPx;
-        segs.forEach(function (seg) {
-            pg.save();
-            var ops = [];
-            if (scale !== 1) ops.push(PDFLib.setCharacterSqueeze(scale * 100));
-            if (seg.face.synthBold) {
-                ops.push(PDFLib.setTextRenderingMode(PDFLib.TextRenderingMode.FillAndOutline));
-                ops.push(PDFLib.setLineWidth(px(spec.sizePx / 28)));
-                ops.push(PDFLib.setStrokingColor(col));
-            }
-            if (ops.length) pg.ops(ops);
-            pg.p.drawText(seg.text, {
-                x: px(cursor), y: pg.y(spec.baselinePx),
-                size: px(spec.sizePx), font: seg.face.font, color: col
-            });
-            pg.restore();
-            cursor += seg.w * scale;
-        });
-        return total * scale;
-    }
-
-    /* drawRuns puts every run on the page at the baseline the browser laid
-       it out on, in the width the browser gave it. */
-    function drawRuns(pg, runs) {
-        runs.forEach(function (r) {
-            var x = r.origin.x + (r.rect.left - r.origin.left);
-            var top = r.origin.y + (r.rect.top - r.origin.top);
-            var lineH = r.rect.bottom - r.rect.top;
-            var domW = r.rect.right - r.rect.left;
-            var glyphH = r.metrics.ascent + r.metrics.descent;
-            var baseline = (lineH - glyphH) / 2 + r.metrics.ascent;
-            var col = parseColor(r.color) || PDFLib.rgb(0, 0, 0);
-            var bg = parseFill(r.background);
-            if (bg) pg.rect(x, top, domW, lineH, { fill: bg.c, fillOpacity: bg.a });
-            // a fragment that starts or ends on a space cannot be fitted to
-            // its rect: the browser collapses those, the measurement does not
-            var w = drawFragment(pg, {
-                text: r.text, names: r.names,
-                bold: r.weight >= 600, italic: r.italic,
-                sizePx: r.size, xPx: x, baselinePx: top + baseline,
-                color: col, fitPx: /^\s|\s$/.test(r.text) ? 0 : domW
-            });
-            if (r.underline || r.strike) {
-                var yOff = r.underline ? baseline + r.size * 0.11 : baseline - r.size * 0.28;
-                pg.rect(x, top + yOff, w || domW, Math.max(0.7, r.size * 0.06), { fill: col });
-            }
-        });
-    }
-
-    /* ---------------- rasterizing one element ---------------- */
-
-    /* The documented last resort, and it has one rule: the pixels come out
-       of a render of the whole slide, then get cropped to the element that
-       needed them.
-
-       Rasterizing an element on its own looks tempting and is wrong.
-       html2canvas re-renders a *clone*, and a clone torn out of its
-       absolutely positioned parent loses the width it was laid out in -
-       so the text re-wraps and the export stops matching the editor,
-       which is the one thing it must not do. Rendering the slide keeps
-       every element in the context it was measured in.
-
-       What lands in the PDF is still only that element's own box: one
-       image, at its own position, with everything else on the page a real
-       PDF object. */
-    /* The picture is taken through an SVG <foreignObject>, which means the
-       *browser* lays the element out and paints it - the same engine, the
-       same fonts, the same line breaks as the editor.
-
-       This is not the obvious choice, so: html2canvas was tried first and
-       is wrong for this. It re-implements layout over a clone, and on
-       mixed CJK/Latin text with pre-wrap it breaks lines somewhere else
-       than the browser did, which is precisely the failure this whole
-       rework exists to remove. A foreignObject cannot re-wrap anything,
-       because it is not re-laying anything out.
-
-       The cost is that computed styles have to be inlined onto the clone
-       (an SVG image cannot reach the page's stylesheets) and every
-       resource must already be a data URL - which, in a slide, it is. */
-    function rasterizeElement(el, w, h) {
-        try {
-            var clone = el.cloneNode(true);
-            inlineStyles(el, clone);
-            // the foreignObject supplies the box, so the clone must not
-            // carry the absolute placement it had on the slide
-            clone.style.position = "static";
-            clone.style.left = "auto";
-            clone.style.top = "auto";
-            clone.style.transform = "none";
-            clone.style.margin = "0";
-            clone.style.width = w + "px";
-            clone.style.height = h + "px";
-
-            var cw = Math.max(1, Math.ceil(w)), ch = Math.max(1, Math.ceil(h));
-            /* The markup inside a foreignObject has to be well-formed XML,
-               and innerHTML is not: it writes <br> unclosed, which makes the
-               whole SVG fail to parse and the element vanish from the page.
-               XMLSerializer writes real XHTML, so it cannot. */
-            var wrap = document.createElementNS("http://www.w3.org/1999/xhtml", "div");
-            wrap.setAttribute("style", "width:" + cw + "px;height:" + ch + "px;");
-            wrap.appendChild(clone);
-            var xhtml = new XMLSerializer().serializeToString(wrap);
-            var svg = '<svg xmlns="http://www.w3.org/2000/svg" width="' + cw +
-                '" height="' + ch + '"><foreignObject x="0" y="0" width="' + cw +
-                '" height="' + ch + '">' + xhtml + "</foreignObject></svg>";
-            var url = "data:image/svg+xml;charset=utf-8," + encodeURIComponent(svg);
-            return new Promise(function (resolve) {
-                var img = new Image();
-                img.onload = function () {
-                    try {
-                        var c = document.createElement("canvas");
-                        c.width = Math.max(1, Math.round(cw * RASTER_SCALE));
-                        c.height = Math.max(1, Math.round(ch * RASTER_SCALE));
-                        var g = c.getContext("2d");
-                        g.drawImage(img, 0, 0, c.width, c.height);
-                        resolve(c.toDataURL("image/png"));
-                    } catch (e) { resolve(null); }
-                };
-                img.onerror = function () { resolve(null); };
-                img.src = url;
-            });
-        } catch (e) {
-            return Promise.resolve(null);
-        }
-    }
-
-    /* rasterizeFallback is the very last resort, for the case where even
-       the foreignObject route fails. html2canvas re-implements layout and
-       can break mixed-script lines somewhere the browser did not, so it is
-       only ever reached when the alternative is dropping the element
-       from the page entirely - which would be worse. */
-    function rasterizeFallback(el, w, h) {
-        if (typeof html2canvas === "undefined") return Promise.resolve(null);
-        return html2canvas(el, {
-            scale: RASTER_SCALE, useCORS: true, backgroundColor: null, logging: false
-        }).then(function (canvas) {
-            return canvas.toDataURL("image/png");
-        }).catch(function () { return null; });
-    }
-
-    /* inlineStyles copies the computed style of every node in the subtree
-       onto the clone, because the SVG image has no access to the page's
-       stylesheets. Slide markup is small - a few divs and spans - so
-       walking the whole property list is affordable and leaves nothing out. */
-    function inlineStyles(src, dst) {
-        var cs = window.getComputedStyle(src);
-        var out = "";
-        for (var i = 0; i < cs.length; i++) {
-            var prop = cs[i];
-            out += prop + ":" + cs.getPropertyValue(prop) + ";";
-        }
-        dst.setAttribute("style", out);
-        var a = src.children, b = dst.children;
-        for (var j = 0; j < a.length && j < b.length; j++) inlineStyles(a[j], b[j]);
-    }
-
     // boxOf returns an element's box in slide coordinates
     // boxOf returns an element's box in slide coordinates
     function boxOf(el, stageEl) {
     function boxOf(el, stageEl) {
         var r = el.getBoundingClientRect(), s = stageEl.getBoundingClientRect();
         var r = el.getBoundingClientRect(), s = stageEl.getBoundingClientRect();
@@ -890,69 +222,6 @@ var SlidesPdf = (function () {
         });
         });
     }
     }
 
 
-    /* ---------------- images ---------------- */
-
-    /* filteredImageData re-encodes a picture through a canvas when it
-       carries a colour treatment. Applying a filter is a pixel operation in
-       any renderer, so this is the correct way to do it, not a fallback. */
-    function filteredImageData(src, filter, naturalW, naturalH) {
-        return new Promise(function (resolve) {
-            var img = new Image();
-            img.onload = function () {
-                try {
-                    var c = document.createElement("canvas");
-                    c.width = naturalW || img.naturalWidth;
-                    c.height = naturalH || img.naturalHeight;
-                    var g = c.getContext("2d");
-                    g.filter = filter;
-                    g.drawImage(img, 0, 0, c.width, c.height);
-                    resolve(c.toDataURL("image/png"));
-                } catch (e) { resolve(null); }
-            };
-            img.onerror = function () { resolve(null); };
-            img.src = src;
-        });
-    }
-
-    /* sniffImage decides which embedder to use from the bytes themselves
-       rather than from a mime string, which a fetched file may not carry */
-    function sniffImage(bytes) {
-        if (bytes.length > 3 && bytes[0] === 0xFF && bytes[1] === 0xD8) return "jpg";
-        return "png";
-    }
-
-    /* embedImage caches by source string: a deck that uses one picture on
-       twenty slides embeds its bytes once.
-
-       A native .ppta keeps its large media out of the body as media?file=
-       links, so a source that is not a data URL is fetched. Embedding the
-       original bytes is the point - re-encoding through a canvas would
-       turn a photo into a much larger lossless PNG. */
-    function makeImageEmbedder(pdfDoc) {
-        var cache = {};
-        function embedBytes(bytes) {
-            return sniffImage(bytes) === "jpg" ? pdfDoc.embedJpg(bytes) : pdfDoc.embedPng(bytes);
-        }
-        return function (src) {
-            if (!src) return Promise.resolve(null);
-            if (cache[src]) return cache[src];
-            var p;
-            var d = dataUrlBytes(src);
-            if (d) {
-                p = /jpe?g/i.test(d.mime) ? pdfDoc.embedJpg(d.bytes) : embedBytes(d.bytes);
-            } else {
-                p = fetch(src).then(function (r) {
-                    if (!r.ok) throw new Error("cannot read " + src);
-                    return r.arrayBuffer();
-                }).then(function (buf) {
-                    return embedBytes(new Uint8Array(buf));
-                });
-            }
-            cache[src] = p.catch(function () { return null; });
-            return cache[src];
-        };
-    }
-
     /* ---------------- SVG (charts) -> PDF vectors ---------------- */
     /* ---------------- SVG (charts) -> PDF vectors ---------------- */
 
 
     /* OfficeCharts draws with rect / line / polyline / polygon / path /
     /* OfficeCharts draws with rect / line / polyline / polygon / path /

Some files were not shown because too many files changed in this diff