Files
kjol/go/webui/pdf.go

581 lines
19 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package webui
// A minimal PDF writer — exactly enough of the spec to paginate a table, and no
// more. Stdlib only (the gowasm engine takes no third-party dependency), which is
// also why the TSX's pdf-lib could not simply be swapped for a Go port of it.
//
// What is implemented:
//
// - Document structure: header, indirect objects, a classic cross-reference
// TABLE (not an xref stream), trailer, startxref, %%EOF.
// - Pages in portrait or landscape, US Letter (612x792 pt) — the same page size
// the TSX used (pdf-lib's default), so exports look identical.
// - Text in the two standard Type1 fonts that need no embedding: Helvetica and
// Helvetica-Bold, in WinAnsiEncoding.
// - Text measurement from the Adobe AFM glyph-width tables (below). Without
// widths you cannot size columns, truncate to fit, or right-align a number.
// - Filled rectangles and stroked lines (the grid, the zebra band, the rules).
//
// What is deliberately NOT implemented: images (so no logo — see
// autotable_export.go), compression (streams are plain, which makes the output
// greppable and the tests meaningful), transparency, annotations, outlines,
// metadata, encryption, and any font that needs embedding.
//
// Coordinates are PDF user space: origin bottom-left, y grows UP, units are
// points (1/72"). A caller lays a table out top-down by starting at
// height-margin and subtracting.
import (
"bytes"
"fmt"
"strconv"
"strings"
)
// PDFOrientation mirrors the TSX's PDFOrientation union (landscape is 0, and so
// is Go's zero value — landscape is the default for a table, as it was there).
type PDFOrientation int
const (
PDF_ORIENTATION_LANDSCAPE PDFOrientation = 0
PDF_ORIENTATION_PORTRAIT PDFOrientation = 1
)
// US Letter, in points. Portrait is 612x792; landscape swaps them.
const (
pdfLetterShort = 612.0
pdfLetterLong = 792.0
)
// PDFColor is an RGB fill/stroke color, each component in [0,1].
type PDFColor struct{ R, G, B float64 }
// PDFGray is the shade of gray at v (0 = black, 1 = white).
func PDFGray(v float64) PDFColor { return PDFColor{v, v, v} }
// PDFTextStyle is how a run of text is drawn.
type PDFTextStyle struct {
Size float64 // in points; 0 means 10
Bold bool // Helvetica-Bold instead of Helvetica
Color PDFColor
}
// PDFOptions configures a new document.
type PDFOptions struct {
Orientation PDFOrientation
// Width/Height override the page size in points. Zero means US Letter in the
// chosen orientation.
Width, Height float64
}
type pdfPage struct {
content bytes.Buffer
}
// PDF is a document being written. Draw onto the current page (the one AddPage
// last created); SetPage rewinds to an earlier one, which is how a footer like
// "Page 2 of 7" gets stamped onto pages that were finished before the total was
// known.
type PDF struct {
w, h float64
pages []*pdfPage
cur int
}
// NewPDF creates an empty document. It has no pages until AddPage is called;
// Bytes on a page-less document emits one blank page, because a PDF with zero
// pages is not a valid PDF.
func NewPDF(o PDFOptions) *PDF {
w, h := pdfLetterLong, pdfLetterShort // landscape
if o.Orientation == PDF_ORIENTATION_PORTRAIT {
w, h = pdfLetterShort, pdfLetterLong
}
if o.Width > 0 {
w = o.Width
}
if o.Height > 0 {
h = o.Height
}
return &PDF{w: w, h: h, cur: -1}
}
// Width / Height are the page size in points.
func (p *PDF) Width() float64 { return p.w }
func (p *PDF) Height() float64 { return p.h }
// PageCount is how many pages exist so far.
func (p *PDF) PageCount() int { return len(p.pages) }
// AddPage appends a blank page and makes it current.
func (p *PDF) AddPage() {
p.pages = append(p.pages, &pdfPage{})
p.cur = len(p.pages) - 1
}
// SetPage makes page i (0-based) current, so a later pass can draw on it. Out of
// range is a no-op — a footer loop must never take the app down.
func (p *PDF) SetPage(i int) {
if i >= 0 && i < len(p.pages) {
p.cur = i
}
}
// page returns the current page, creating one if the caller drew before adding.
func (p *PDF) page() *pdfPage {
if p.cur < 0 || p.cur >= len(p.pages) {
p.AddPage()
}
return p.pages[p.cur]
}
// Text draws s with its left edge at x and its BASELINE at y.
func (p *PDF) Text(x, y float64, s string, st PDFTextStyle) {
if s == "" {
return
}
size := st.Size
if size <= 0 {
size = 10
}
font := "/F1"
if st.Bold {
font = "/F2"
}
c := &p.page().content
fmt.Fprintf(c, "BT\n%s %s Tf\n%s %s %s rg\n1 0 0 1 %s %s Tm\n%s Tj\nET\n",
font, pdfNum(size),
pdfNum(st.Color.R), pdfNum(st.Color.G), pdfNum(st.Color.B),
pdfNum(x), pdfNum(y),
pdfString(s),
)
}
// TextRight draws s with its RIGHT edge at x — how a number lands under the right
// edge of its column, and the only reason the width table has to be correct.
func (p *PDF) TextRight(x, y float64, s string, st PDFTextStyle) {
p.Text(x-PDFTextWidth(s, styleSize(st), st.Bold), y, s, st)
}
// TextCenter draws s centered on x.
func (p *PDF) TextCenter(x, y float64, s string, st PDFTextStyle) {
p.Text(x-PDFTextWidth(s, styleSize(st), st.Bold)/2, y, s, st)
}
func styleSize(st PDFTextStyle) float64 {
if st.Size <= 0 {
return 10
}
return st.Size
}
// Line strokes a straight line.
func (p *PDF) Line(x1, y1, x2, y2, thickness float64, color PDFColor) {
if thickness <= 0 {
thickness = 1
}
fmt.Fprintf(&p.page().content, "%s %s %s RG\n%s w\n%s %s m\n%s %s l\nS\n",
pdfNum(color.R), pdfNum(color.G), pdfNum(color.B),
pdfNum(thickness),
pdfNum(x1), pdfNum(y1), pdfNum(x2), pdfNum(y2),
)
}
// Rect fills a rectangle whose lower-left corner is (x,y). There is no stroked
// variant: a table's borders are drawn as lines, so nothing needs one.
func (p *PDF) Rect(x, y, w, h float64, color PDFColor) {
if w <= 0 || h <= 0 {
return
}
fmt.Fprintf(&p.page().content, "%s %s %s rg\n%s %s %s %s re\nf\n",
pdfNum(color.R), pdfNum(color.G), pdfNum(color.B),
pdfNum(x), pdfNum(y), pdfNum(w), pdfNum(h),
)
}
// ---- serialization ----
// Object layout is fixed, which is what keeps the xref arithmetic honest:
//
// 1 Catalog
// 2 Pages
// 3 Helvetica (/F1)
// 4 Helvetica-Bold (/F2)
// 5, 6 page 0: the page object, then its content stream
// 7, 8 page 1: …
const (
pdfObjCatalog = 1
pdfObjPages = 2
pdfObjFont = 3
pdfObjFontBold = 4
)
// pageObjNum / contentObjNum are the object numbers for page i.
func pageObjNum(i int) int { return 5 + 2*i }
func contentObjNum(i int) int { return 6 + 2*i }
// Bytes serializes the document. The result is a complete, standalone PDF file.
func (p *PDF) Bytes() []byte {
pages := p.pages
if len(pages) == 0 {
pages = []*pdfPage{{}} // a zero-page PDF is invalid; ship one blank page
}
nObjs := 4 + 2*len(pages)
var buf bytes.Buffer
buf.WriteString("%PDF-1.4\n")
// The conventional binary marker: four bytes >127 on a comment line, so a tool
// that sniffs the first lines classifies the file as binary and does not
// helpfully mangle its line endings.
buf.Write([]byte{'%', 0xE2, 0xE3, 0xCF, 0xD3, '\n'})
// offsets[n] is the byte offset of object n (index 0 unused).
offsets := make([]int, nObjs+1)
obj := func(n int, body string) {
offsets[n] = buf.Len()
fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", n, body)
}
stream := func(n int, data []byte) {
offsets[n] = buf.Len()
fmt.Fprintf(&buf, "%d 0 obj\n<< /Length %d >>\nstream\n", n, len(data))
buf.Write(data)
buf.WriteString("\nendstream\nendobj\n")
}
obj(pdfObjCatalog, fmt.Sprintf("<< /Type /Catalog /Pages %d 0 R >>", pdfObjPages))
var kids strings.Builder
for i := range pages {
if i > 0 {
kids.WriteByte(' ')
}
fmt.Fprintf(&kids, "%d 0 R", pageObjNum(i))
}
obj(pdfObjPages, fmt.Sprintf("<< /Type /Pages /Kids [%s] /Count %d >>", kids.String(), len(pages)))
obj(pdfObjFont, "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>")
obj(pdfObjFontBold, "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding >>")
for i, pg := range pages {
obj(pageObjNum(i), fmt.Sprintf(
"<< /Type /Page /Parent %d 0 R /MediaBox [0 0 %s %s] "+
"/Resources << /Font << /F1 %d 0 R /F2 %d 0 R >> >> /Contents %d 0 R >>",
pdfObjPages, pdfNum(p.w), pdfNum(p.h),
pdfObjFont, pdfObjFontBold, contentObjNum(i),
))
stream(contentObjNum(i), pg.content.Bytes())
}
// The cross-reference table. Every entry is exactly 20 bytes — 10-digit
// offset, space, 5-digit generation, space, type, and a two-byte EOL — because
// readers index into it arithmetically rather than parsing it.
xref := buf.Len()
fmt.Fprintf(&buf, "xref\n0 %d\n", nObjs+1)
buf.WriteString("0000000000 65535 f \n")
for n := 1; n <= nObjs; n++ {
fmt.Fprintf(&buf, "%010d 00000 n \n", offsets[n])
}
fmt.Fprintf(&buf, "trailer\n<< /Size %d /Root %d 0 R >>\nstartxref\n%d\n%%%%EOF\n",
nObjs+1, pdfObjCatalog, xref)
return buf.Bytes()
}
// pdfNum formats a coordinate. PDF has no exponent notation, so strconv's 'g'/-1
// shortest form is unusable ("1e-07" is a syntax error in a content stream); and
// three decimals is well past what a 72dpi page can resolve.
func pdfNum(v float64) string {
if v == 0 { // also collapses -0
return "0"
}
s := strconv.FormatFloat(v, 'f', 3, 64)
s = strings.TrimRight(s, "0")
s = strings.TrimSuffix(s, ".")
if s == "" || s == "-" {
return "0"
}
return s
}
// pdfString renders s as a PDF literal string, transcoded to WinAnsi. Anything a
// parser could choke on — the delimiters, the escape character, every byte
// outside printable ASCII — is escaped, the last as three-digit octal.
func pdfString(s string) string {
var b strings.Builder
b.WriteByte('(')
for _, r := range s {
c := winAnsiByte(r)
switch {
case c == '(' || c == ')' || c == '\\':
b.WriteByte('\\')
b.WriteByte(c)
case c < 32 || c > 126:
fmt.Fprintf(&b, "\\%03o", c)
default:
b.WriteByte(c)
}
}
b.WriteByte(')')
return b.String()
}
// ---- encoding ----
// winAnsiByte maps a rune to its WinAnsiEncoding code. Latin-1 passes straight
// through; the printer's punctuation Windows squats in 0x80-0x9F (curly quotes,
// dashes, the ellipsis the truncator appends) is mapped explicitly. Anything else
// — CJK, emoji, a tab in the middle of a cell — becomes '?', because the standard
// 14 fonts have no glyph for it and silently dropping it would misalign the
// column instead.
func winAnsiByte(r rune) byte {
switch {
case r == '\t' || r == '\n' || r == '\r':
return ' '
case r >= 32 && r <= 126:
return byte(r)
case r >= 0xA0 && r <= 0xFF:
return byte(r)
}
if c, ok := winAnsiSpecial[r]; ok {
return c
}
return '?'
}
var winAnsiSpecial = map[rune]byte{
'€': 0x80, // euro
'': 0x82, // single low quote
'ƒ': 0x83, // florin
'„': 0x84, // double low quote
'…': 0x85, // ellipsis <- the truncation marker
'†': 0x86, // dagger
'‡': 0x87, // double dagger
'ˆ': 0x88, // circumflex
'‰': 0x89, // per mille
'Š': 0x8A, // S caron
'': 0x8B, // single left guillemet
'Œ': 0x8C, // OE
'Ž': 0x8E, // Z caron
'': 0x91, // left single quote
'': 0x92, // right single quote (apostrophe)
'“': 0x93, // left double quote
'”': 0x94, // right double quote
'•': 0x95, // bullet
'': 0x96, // en dash
'—': 0x97, // em dash
'˜': 0x98, // small tilde
'™': 0x99, // trademark
'š': 0x9A, // s caron
'': 0x9B, // single right guillemet
'œ': 0x9C, // oe
'ž': 0x9E, // z caron
'Ÿ': 0x9F, // Y dieresis
}
// ---- measurement ----
// PDFTextWidth is the width of s in points, at the given size, in Helvetica (or
// Helvetica-Bold). Widths come from the Adobe AFM tables, in 1/1000 em.
func PDFTextWidth(s string, size float64, bold bool) float64 {
table := &helveticaWidths
if bold {
table = &helveticaBoldWidths
}
total := 0.0
for _, r := range s {
total += float64(table[winAnsiByte(r)])
}
return total * size / 1000
}
// PDFTruncate shortens s until it fits maxWidth, appending an ellipsis — the
// ellipsis being measured too, so the result really does fit. Returns "" when not
// even the ellipsis fits, which is the honest answer for a column that narrow.
func PDFTruncate(s string, maxWidth, size float64, bold bool) string {
if s == "" || PDFTextWidth(s, size, bold) <= maxWidth {
return s
}
runes := []rune(s)
for n := len(runes) - 1; n > 0; n-- {
if PDFTextWidth(string(runes[:n])+"…", size, bold) <= maxWidth {
return string(runes[:n]) + "…"
}
}
if PDFTextWidth("…", size, bold) <= maxWidth {
return "…"
}
return ""
}
// The Helvetica / Helvetica-Bold advance widths, indexed by WinAnsi code, in
// 1/1000 em — the Adobe Core-14 AFM data, which is what "no font embedding
// required" actually costs you: the widths have to live somewhere, and this is
// where.
//
// Codes 32-126 are the exact AFM values (note 39 is quotesingle and 96 is grave
// under WinAnsi, NOT quoteright/quoteleft as under StandardEncoding — a classic
// off-by-one-glyph). 0x80-0xFF are the AFM values for the Latin-1 and printer's
// punctuation glyphs; the accented letters carry the advance width of their base
// letter, which is exactly how the composite glyphs are built. Unmapped codes
// (0x81, 0x8D, 0x8F, 0x90, 0x9D) fall back to the width of 'n', so a stray byte
// costs a plausible amount of space rather than zero.
var (
helveticaWidths [256]uint16
helveticaBoldWidths [256]uint16
)
// helveticaAscii/helveticaBoldAscii are codes 32..126, in order.
var helveticaAscii = [95]uint16{
278, 278, 355, 556, 556, 889, 667, 191, 333, 333, // 32-41 space ! " # $ % & ' ( )
389, 584, 278, 333, 278, 278, 556, 556, 556, 556, // 42-51 * + , - . / 0 1 2 3
556, 556, 556, 556, 556, 556, 278, 278, 584, 584, // 52-61 4 5 6 7 8 9 : ; < =
584, 556, 1015, 667, 667, 722, 722, 667, 611, 778, // 62-71 > ? @ A B C D E F G
722, 278, 500, 667, 556, 833, 722, 778, 667, 778, // 72-81 H I J K L M N O P Q
722, 667, 611, 722, 667, 944, 667, 667, 611, 278, // 82-91 R S T U V W X Y Z [
278, 278, 469, 556, 333, 556, 556, 500, 556, 556, // 92-101 \ ] ^ _ ` a b c d e
278, 556, 556, 222, 222, 500, 222, 833, 556, 556, // 102-111 f g h i j k l m n o
556, 556, 333, 500, 278, 556, 500, 722, 500, 500, // 112-121 p q r s t u v w x y
500, 334, 260, 334, 584, // 122-126 z { | } ~
}
var helveticaBoldAscii = [95]uint16{
278, 333, 474, 556, 556, 889, 722, 238, 333, 333, // 32-41
389, 584, 278, 333, 278, 278, 556, 556, 556, 556, // 42-51
556, 556, 556, 556, 556, 556, 333, 333, 584, 584, // 52-61
584, 611, 975, 722, 722, 722, 722, 667, 611, 778, // 62-71
722, 278, 556, 722, 611, 833, 722, 778, 667, 778, // 72-81
722, 667, 611, 722, 667, 944, 667, 667, 611, 333, // 82-91
278, 333, 584, 556, 333, 556, 611, 556, 611, 556, // 92-101
333, 611, 611, 278, 278, 556, 278, 889, 611, 611, // 102-111
611, 611, 389, 556, 333, 611, 556, 778, 556, 556, // 112-121
500, 389, 280, 389, 584, // 122-126
}
// The high half: {code: {regular, bold}}.
var helveticaHigh = map[byte][2]uint16{
0x80: {556, 556}, // euro
0x82: {222, 278}, // quotesinglbase
0x83: {556, 556}, // florin
0x84: {333, 500}, // quotedblbase
0x85: {1000, 1000}, // ellipsis
0x86: {556, 556}, // dagger
0x87: {556, 556}, // daggerdbl
0x88: {333, 333}, // circumflex
0x89: {1000, 1000}, // perthousand
0x8A: {667, 667}, // Scaron
0x8B: {333, 333}, // guilsinglleft
0x8C: {1000, 1000}, // OE
0x8E: {611, 611}, // Zcaron
0x91: {222, 278}, // quoteleft
0x92: {222, 278}, // quoteright
0x93: {333, 500}, // quotedblleft
0x94: {333, 500}, // quotedblright
0x95: {350, 350}, // bullet
0x96: {556, 556}, // endash
0x97: {1000, 1000}, // emdash
0x98: {333, 333}, // tilde
0x99: {1000, 1000}, // trademark
0x9A: {500, 556}, // scaron
0x9B: {333, 333}, // guilsinglright
0x9C: {944, 944}, // oe
0x9E: {500, 500}, // zcaron
0x9F: {667, 667}, // Ydieresis
0xA0: {278, 278}, // nbsp
0xA1: {333, 333}, // exclamdown
0xA2: {556, 556}, // cent
0xA3: {556, 556}, // sterling
0xA4: {556, 556}, // currency
0xA5: {556, 556}, // yen
0xA6: {260, 280}, // brokenbar
0xA7: {556, 556}, // section
0xA8: {333, 333}, // dieresis
0xA9: {737, 737}, // copyright
0xAA: {370, 370}, // ordfeminine
0xAB: {556, 556}, // guillemotleft
0xAC: {584, 584}, // logicalnot
0xAD: {333, 333}, // soft hyphen
0xAE: {737, 737}, // registered
0xAF: {333, 333}, // macron
0xB0: {400, 400}, // degree
0xB1: {584, 584}, // plusminus
0xB2: {333, 333}, // twosuperior
0xB3: {333, 333}, // threesuperior
0xB4: {333, 333}, // acute
0xB5: {556, 611}, // mu
0xB6: {537, 556}, // paragraph
0xB7: {278, 278}, // periodcentered
0xB8: {333, 333}, // cedilla
0xB9: {333, 333}, // onesuperior
0xBA: {365, 365}, // ordmasculine
0xBB: {556, 556}, // guillemotright
0xBC: {834, 834}, // onequarter
0xBD: {834, 834}, // onehalf
0xBE: {834, 834}, // threequarters
0xBF: {611, 611}, // questiondown
0xC6: {1000, 1000}, // AE
0xD0: {722, 722}, // Eth
0xD7: {584, 584}, // multiply
0xD8: {778, 778}, // Oslash
0xDD: {667, 667}, // Yacute
0xDE: {667, 667}, // Thorn
0xDF: {611, 611}, // germandbls
0xE6: {889, 889}, // ae
0xF0: {556, 611}, // eth
0xF7: {584, 584}, // divide
0xF8: {611, 611}, // oslash
0xFD: {500, 556}, // yacute
0xFE: {556, 611}, // thorn
0xFF: {500, 556}, // ydieresis
}
// accentBase maps the composite Latin-1 letters to the ASCII letter whose advance
// width they share (Helvetica builds them by stacking an accent over the base
// glyph, which does not widen it). The accented i's are the exception: they are
// built over dotlessi, which is wider than i.
var accentBase = map[byte]byte{
0xC0: 'A', 0xC1: 'A', 0xC2: 'A', 0xC3: 'A', 0xC4: 'A', 0xC5: 'A',
0xC7: 'C',
0xC8: 'E', 0xC9: 'E', 0xCA: 'E', 0xCB: 'E',
0xD1: 'N',
0xD2: 'O', 0xD3: 'O', 0xD4: 'O', 0xD5: 'O', 0xD6: 'O',
0xD9: 'U', 0xDA: 'U', 0xDB: 'U', 0xDC: 'U',
0xE0: 'a', 0xE1: 'a', 0xE2: 'a', 0xE3: 'a', 0xE4: 'a', 0xE5: 'a',
0xE7: 'c',
0xE8: 'e', 0xE9: 'e', 0xEA: 'e', 0xEB: 'e',
0xF1: 'n',
0xF2: 'o', 0xF3: 'o', 0xF4: 'o', 0xF5: 'o', 0xF6: 'o',
0xF9: 'u', 0xFA: 'u', 0xFB: 'u', 0xFC: 'u',
}
func init() {
// dotlessi's advance — what the accented i's (0xCC-0xCF, 0xEC-0xEF) are built on.
const dotlessI, dotlessIBold = 278, 278
fill := func(dst *[256]uint16, ascii *[95]uint16, pick int, dotless uint16) {
fallback := ascii['n'-32]
for i := range dst {
dst[i] = fallback
}
for i, w := range ascii {
dst[32+i] = w
}
for c, w := range helveticaHigh {
dst[c] = w[pick]
}
for c, base := range accentBase {
dst[c] = ascii[base-32]
}
for _, c := range []byte{0xCC, 0xCD, 0xCE, 0xCF, 0xEC, 0xED, 0xEE, 0xEF} {
dst[c] = dotless
}
// Codes 0-31 are unprintable and never emitted (winAnsiByte folds
// whitespace to a space), but give them the space width anyway so a
// measurement can never be wildly off.
for i := 0; i < 32; i++ {
dst[i] = ascii[0]
}
}
fill(&helveticaWidths, &helveticaAscii, 0, dotlessI)
fill(&helveticaBoldWidths, &helveticaBoldAscii, 1, dotlessIBold)
}