Re-encode plates to lossless JBIG2 / JPEG at build time (115 MB -> 12 MB)

The scan is a bilevel 300 dpi JBIG2 layer plus lower-resolution colour, so
storing every plate as 8-bit PNG only added size. tools/optimize_plates.py
writes copies under work/plates-opt (bilevel to generic-mode, lossless JBIG2;
genuine halftone to JPEG q88) and repoints work/pages at them; plates/ and
src/ are untouched. The Dockerfile builds jbig2enc and adds qpdf, ghostscript
and img2pdf.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-29 14:36:18 +06:00
co-authored by Claude Sonnet 5.5
parent c4aec3becf
commit d48ce1f50c
4 changed files with 213 additions and 1 deletions
+14
View File
@@ -9,9 +9,23 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
poppler-utils tesseract-ocr tesseract-ocr-eng \
python3 python3-pip python3-pil python3-numpy \
make curl ca-certificates \
qpdf ghostscript img2pdf \
git automake autoconf libtool g++ pkg-config libleptonica-dev zlib1g-dev \
&& rm -rf /var/lib/apt/lists/* \
&& pip3 install --no-cache-dir --break-system-packages pdfplumber fonttools \
&& fc-cache -f
# jbig2enc: the encoder the ProQuest scan itself uses for its bilevel page
# layer, and the reason a 2556x3305 page fits in 4900 bytes there. It builds a
# dictionary of repeated letterforms and stores each shape once, which general
# compressors cannot do -- measured on this scan, 0.019 bits/pixel against
# 0.084 for CCITT G4 and 0.159 for 1-bit PNG. Not packaged for Ubuntu 24.04, so
# it is built from source.
RUN git clone --depth 1 https://github.com/agl/jbig2enc /tmp/jbig2enc \
&& cd /tmp/jbig2enc \
&& ./autogen.sh && ./configure && make -j"$(nproc)" && make install \
&& ldconfig \
&& rm -rf /tmp/jbig2enc
WORKDIR /work
# The project directory is bind-mounted at /work (see compose.yaml); nothing is copied in.
CMD ["bash"]
+1 -1
View File
@@ -7,7 +7,7 @@ export GID := $(shell id -g)
RUN := $(if $(NODOCKER),,docker compose run --rm -T tex)
OUT := ross-1988-retypeset.pdf
SOURCES := build.sh tools/polish.py src/preamble.tex src/colophon.tex src/errata.tex \
SOURCES := build.sh tools/polish.py tools/optimize_plates.py src/preamble.tex src/colophon.tex src/errata.tex \
$(wildcard src/pages/*.tex)
.DEFAULT_GOAL := pdf
+6
View File
@@ -21,6 +21,12 @@ trap 'rmdir work/.buildlock 2>/dev/null' EXIT INT TERM
# Three full passes run below regardless, so a stale aux buys nothing.
rm -f work/main.aux work/main.toc work/main.out
python3 tools/polish.py >/dev/null # typographic pass: src/pages -> work/pages
# Re-encode the plates to match what they actually hold: bilevel type goes to
# lossless JBIG2, genuine halftone stays 8-bit as JPEG. Writes copies under
# work/plates-opt and repoints work/pages at them -- plates/ and src/ are
# never touched, so deleting work/ reverts it. Must run AFTER polish.py, which
# is what creates work/pages.
python3 tools/optimize_plates.py
{ cat src/preamble.tex
printf '%s\n' '\begin{document}\immediate\openout\unsurefile=unsure.log\immediate\openout\erratafile=errata.log\immediate\openout\qslipfile=qslips.log'
printf '\\input{%s}\n' work/pages/p0001.tex src/colophon.tex
+192
View File
@@ -0,0 +1,192 @@
#!/usr/bin/env python3
r"""Re-encode the plate images with codecs matched to what they actually hold.
WHY
The ProQuest scan stores each page as MRC: a 300 dpi bilevel JBIG2 text layer
plus a 150/75 dpi JPX wash for paper tint. Our plates are cropped from a 300 dpi
re-rasterisation of that composite and saved as 8-bit greyscale PNG, which
stores, per pixel, about 60x what the source spends -- almost all of it
antialiasing our own render introduced and a background tint upsampled from
150 dpi. Measured on this scan:
source JBIG2 0.019 bits/pixel
CCITT G4 0.084
1-bit PNG 0.159
our 8-bit PNG ~1.2
The originals in plates/ are never modified. This writes re-encoded copies to
work/plates-opt/ and repoints work/pages/*.tex at them, so the transcription in
src/ is untouched and the whole thing reverts by deleting work/.
Safe because \plateop and \plate scale plates to fit the measure
(width=\hsize, keepaspectratio) -- pixel dimensions do not affect layout. The
inline specimens are NOT touched: \ig sets them at natural size (1px = 1/300in),
so their resolution is load-bearing, and they are only 1.6 MB in total.
CLASSIFICATION
A plate whose midtone fraction is low is type on paper: its real content is the
bilevel layer, and thresholding discards only our own resampling. A plate with
substantial midtone is a photograph or halftone, where grey IS the content;
those are kept 8-bit and JPEG-encoded instead.
"""
import argparse
import glob
import os
import re
import shutil
import subprocess
import sys
import tempfile
import numpy as np
from PIL import Image
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
OUTDIR = os.path.join(ROOT, 'work', 'plates-opt')
def midtone(a):
h = np.bincount(a.ravel(), minlength=256)
return 1.0 - (h[:40].sum() + h[216:].sum()) / a.size
def _pdf_escape(n):
return str(n).encode()
def jbig2_pdf(png_path, out_pdf):
"""Encode bilevel with jbig2enc in GENERIC (lossless) mode and wrap the
stream in a one-page PDF.
Generic mode, NOT symbol mode. jbig2enc's `-s` builds a dictionary of
glyph shapes and substitutes visually similar symbols for one another --
it is lossy, and it is the mechanism behind the Xerox scanning affair
where scanned digits were silently swapped. Measured here it altered
1-4% of pixels. In a thesis whose argument IS the shape of the
letterforms, a substituted glyph would be a fabricated reading, so the
saving is not available to us at any price. Generic mode turned out to be
both lossless and slightly SMALLER on these single-page plates (19,749
bytes against 20,831 on p0063), because a per-page symbol dictionary does
not repay its own overhead.
Returns False if the encoder is unavailable, so the caller can fall back.
"""
if shutil.which('jbig2') is None:
return False
with tempfile.TemporaryDirectory() as td:
# -p emits the PDF-EMBEDDED stream form. Without it jbig2enc writes
# the standalone JBIG2 *file* format, header and all, which a PDF
# reader cannot decode -- it silently yields a blank page rather than
# an error, so this only shows up if you check the decoded pixels.
r = subprocess.run(['jbig2', '-p', os.path.abspath(png_path)],
cwd=td, capture_output=True)
if r.returncode != 0 or not r.stdout:
return False
image_data = r.stdout
im = Image.open(png_path)
w, h = im.size
pw, ph = w * 72.0 / 300.0, h * 72.0 / 300.0
objs = []
objs.append(b"<< /Type /Catalog /Pages 2 0 R >>")
objs.append(b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>")
objs.append(("<< /Type /Page /Parent 2 0 R /MediaBox [0 0 %.2f %.2f] "
"/Resources << /XObject << /Im0 4 0 R >> >> /Contents 5 0 R >>"
% (pw, ph)).encode())
objs.append(("<< /Type /XObject /Subtype /Image /Width %d /Height %d "
"/ColorSpace /DeviceGray /BitsPerComponent 1 "
# no /Decode: JBIG2Decode already yields PDF's polarity
# (0 = black). Adding /Decode [1 0] inverts the plate, which
# showed up as a ~90% pixel mismatch against the original.
"/Filter /JBIG2Decode /Length %d >>\nstream\n"
% (w, h, len(image_data))).encode()
+ image_data + b"\nendstream")
content = ("q %.2f 0 0 %.2f 0 0 cm /Im0 Do Q" % (pw, ph)).encode()
objs.append(b"<< /Length " + _pdf_escape(len(content)) + b" >>\nstream\n"
+ content + b"\nendstream")
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, body in enumerate(objs, start=1):
offsets.append(len(out))
out += b"%d 0 obj\n" % i + body + b"\nendobj\n"
xref = len(out)
out += b"xref\n0 %d\n" % (len(objs) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += b"%010d 00000 n \n" % off
out += (b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n"
% (len(objs) + 1, xref))
with open(out_pdf, 'wb') as f:
f.write(bytes(out))
return True
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--tone-quality', type=int, default=88)
ap.add_argument('--tone-threshold', type=float, default=0.25,
help='midtone fraction above which a plate keeps 8-bit grey')
ap.add_argument('--rewrite', default='work/pages',
help='rewrite \\includegraphics paths in these .tex files')
ap.add_argument('--dry-run', action='store_true')
args = ap.parse_args()
os.makedirs(OUTDIR, exist_ok=True)
mapping = {}
n_b = n_t = 0
before = after = 0
used_jbig2 = used_g4 = 0
for src in sorted(glob.glob(os.path.join(ROOT, 'plates', 'p0*.png'))):
if not re.search(r'p\d+\.png$', src):
continue
rel = os.path.relpath(src, ROOT)
a = np.array(Image.open(src).convert('L'))
sz = os.path.getsize(src)
before += sz
stem = os.path.splitext(os.path.basename(src))[0]
if midtone(a) > args.tone_threshold:
dst = os.path.join(OUTDIR, stem + '.jpg')
if not args.dry_run:
Image.fromarray(a).save(dst, 'JPEG', quality=args.tone_quality,
optimize=True)
n_t += 1
else:
dst = os.path.join(OUTDIR, stem + '.pdf')
if not args.dry_run:
bw = os.path.join(OUTDIR, stem + '.bw.png')
Image.fromarray((a > 128)).save(bw, 'PNG', bits=1)
if jbig2_pdf(bw, dst):
used_jbig2 += 1
else:
Image.fromarray((a > 128)).save(dst, 'PDF') # CCITT G4
used_g4 += 1
os.remove(bw)
n_b += 1
if not args.dry_run:
after += os.path.getsize(dst)
mapping[rel] = os.path.relpath(dst, ROOT)
if not args.dry_run and args.rewrite:
pat = re.compile(r'plates/(p\d+)\.png')
changed = 0
for tex in glob.glob(os.path.join(ROOT, args.rewrite, '*.tex')):
s = open(tex).read()
new = pat.sub(lambda m: mapping.get(f'plates/{m.group(1)}.png',
m.group(0)), s)
if new != s:
open(tex, 'w').write(new)
changed += 1
print(f'rewrote {changed} page files to point at work/plates-opt/')
print(f'bilevel {n_b} (jbig2 {used_jbig2}, g4 fallback {used_g4}), tonal {n_t}')
if after:
print(f'plates: {before/1048576:.1f} MB -> {after/1048576:.1f} MB '
f'({after/before:.1%})')
if __name__ == '__main__':
sys.exit(main())