Files
bdeshiandClaude Sonnet 5.5 d48ce1f50c Re-encode plates to lossless JBIG2 / JPEG at build time (115 MB -> 12 MB)
The scan is a bilevel 300 dpi JBIG2 layer plus lower-resolution colour, so
storing every plate as 8-bit PNG only added size. tools/optimize_plates.py
writes copies under work/plates-opt (bilevel to generic-mode, lossless JBIG2;
genuine halftone to JPEG q88) and repoints work/pages at them; plates/ and
src/ are untouched. The Dockerfile builds jbig2enc and adds qpdf, ghostscript
and img2pdf.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-09-29 14:36:18 +06:00

193 lines
7.6 KiB
Python

#!/usr/bin/env python3
r"""Re-encode the plate images with codecs matched to what they actually hold.
WHY
The ProQuest scan stores each page as MRC: a 300 dpi bilevel JBIG2 text layer
plus a 150/75 dpi JPX wash for paper tint. Our plates are cropped from a 300 dpi
re-rasterisation of that composite and saved as 8-bit greyscale PNG, which
stores, per pixel, about 60x what the source spends -- almost all of it
antialiasing our own render introduced and a background tint upsampled from
150 dpi. Measured on this scan:
source JBIG2 0.019 bits/pixel
CCITT G4 0.084
1-bit PNG 0.159
our 8-bit PNG ~1.2
The originals in plates/ are never modified. This writes re-encoded copies to
work/plates-opt/ and repoints work/pages/*.tex at them, so the transcription in
src/ is untouched and the whole thing reverts by deleting work/.
Safe because \plateop and \plate scale plates to fit the measure
(width=\hsize, keepaspectratio) -- pixel dimensions do not affect layout. The
inline specimens are NOT touched: \ig sets them at natural size (1px = 1/300in),
so their resolution is load-bearing, and they are only 1.6 MB in total.
CLASSIFICATION
A plate whose midtone fraction is low is type on paper: its real content is the
bilevel layer, and thresholding discards only our own resampling. A plate with
substantial midtone is a photograph or halftone, where grey IS the content;
those are kept 8-bit and JPEG-encoded instead.
"""
import argparse
import glob
import os
import re
import shutil
import subprocess
import sys
import tempfile
import numpy as np
from PIL import Image
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
OUTDIR = os.path.join(ROOT, 'work', 'plates-opt')
def midtone(a):
h = np.bincount(a.ravel(), minlength=256)
return 1.0 - (h[:40].sum() + h[216:].sum()) / a.size
def _pdf_escape(n):
return str(n).encode()
def jbig2_pdf(png_path, out_pdf):
"""Encode bilevel with jbig2enc in GENERIC (lossless) mode and wrap the
stream in a one-page PDF.
Generic mode, NOT symbol mode. jbig2enc's `-s` builds a dictionary of
glyph shapes and substitutes visually similar symbols for one another --
it is lossy, and it is the mechanism behind the Xerox scanning affair
where scanned digits were silently swapped. Measured here it altered
1-4% of pixels. In a thesis whose argument IS the shape of the
letterforms, a substituted glyph would be a fabricated reading, so the
saving is not available to us at any price. Generic mode turned out to be
both lossless and slightly SMALLER on these single-page plates (19,749
bytes against 20,831 on p0063), because a per-page symbol dictionary does
not repay its own overhead.
Returns False if the encoder is unavailable, so the caller can fall back.
"""
if shutil.which('jbig2') is None:
return False
with tempfile.TemporaryDirectory() as td:
# -p emits the PDF-EMBEDDED stream form. Without it jbig2enc writes
# the standalone JBIG2 *file* format, header and all, which a PDF
# reader cannot decode -- it silently yields a blank page rather than
# an error, so this only shows up if you check the decoded pixels.
r = subprocess.run(['jbig2', '-p', os.path.abspath(png_path)],
cwd=td, capture_output=True)
if r.returncode != 0 or not r.stdout:
return False
image_data = r.stdout
im = Image.open(png_path)
w, h = im.size
pw, ph = w * 72.0 / 300.0, h * 72.0 / 300.0
objs = []
objs.append(b"<< /Type /Catalog /Pages 2 0 R >>")
objs.append(b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>")
objs.append(("<< /Type /Page /Parent 2 0 R /MediaBox [0 0 %.2f %.2f] "
"/Resources << /XObject << /Im0 4 0 R >> >> /Contents 5 0 R >>"
% (pw, ph)).encode())
objs.append(("<< /Type /XObject /Subtype /Image /Width %d /Height %d "
"/ColorSpace /DeviceGray /BitsPerComponent 1 "
# no /Decode: JBIG2Decode already yields PDF's polarity
# (0 = black). Adding /Decode [1 0] inverts the plate, which
# showed up as a ~90% pixel mismatch against the original.
"/Filter /JBIG2Decode /Length %d >>\nstream\n"
% (w, h, len(image_data))).encode()
+ image_data + b"\nendstream")
content = ("q %.2f 0 0 %.2f 0 0 cm /Im0 Do Q" % (pw, ph)).encode()
objs.append(b"<< /Length " + _pdf_escape(len(content)) + b" >>\nstream\n"
+ content + b"\nendstream")
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, body in enumerate(objs, start=1):
offsets.append(len(out))
out += b"%d 0 obj\n" % i + body + b"\nendobj\n"
xref = len(out)
out += b"xref\n0 %d\n" % (len(objs) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += b"%010d 00000 n \n" % off
out += (b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n"
% (len(objs) + 1, xref))
with open(out_pdf, 'wb') as f:
f.write(bytes(out))
return True
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--tone-quality', type=int, default=88)
ap.add_argument('--tone-threshold', type=float, default=0.25,
help='midtone fraction above which a plate keeps 8-bit grey')
ap.add_argument('--rewrite', default='work/pages',
help='rewrite \\includegraphics paths in these .tex files')
ap.add_argument('--dry-run', action='store_true')
args = ap.parse_args()
os.makedirs(OUTDIR, exist_ok=True)
mapping = {}
n_b = n_t = 0
before = after = 0
used_jbig2 = used_g4 = 0
for src in sorted(glob.glob(os.path.join(ROOT, 'plates', 'p0*.png'))):
if not re.search(r'p\d+\.png$', src):
continue
rel = os.path.relpath(src, ROOT)
a = np.array(Image.open(src).convert('L'))
sz = os.path.getsize(src)
before += sz
stem = os.path.splitext(os.path.basename(src))[0]
if midtone(a) > args.tone_threshold:
dst = os.path.join(OUTDIR, stem + '.jpg')
if not args.dry_run:
Image.fromarray(a).save(dst, 'JPEG', quality=args.tone_quality,
optimize=True)
n_t += 1
else:
dst = os.path.join(OUTDIR, stem + '.pdf')
if not args.dry_run:
bw = os.path.join(OUTDIR, stem + '.bw.png')
Image.fromarray((a > 128)).save(bw, 'PNG', bits=1)
if jbig2_pdf(bw, dst):
used_jbig2 += 1
else:
Image.fromarray((a > 128)).save(dst, 'PDF') # CCITT G4
used_g4 += 1
os.remove(bw)
n_b += 1
if not args.dry_run:
after += os.path.getsize(dst)
mapping[rel] = os.path.relpath(dst, ROOT)
if not args.dry_run and args.rewrite:
pat = re.compile(r'plates/(p\d+)\.png')
changed = 0
for tex in glob.glob(os.path.join(ROOT, args.rewrite, '*.tex')):
s = open(tex).read()
new = pat.sub(lambda m: mapping.get(f'plates/{m.group(1)}.png',
m.group(0)), s)
if new != s:
open(tex, 'w').write(new)
changed += 1
print(f'rewrote {changed} page files to point at work/plates-opt/')
print(f'bilevel {n_b} (jbig2 {used_jbig2}, g4 fallback {used_g4}), tonal {n_t}')
if after:
print(f'plates: {before/1048576:.1f} MB -> {after/1048576:.1f} MB '
f'({after/before:.1%})')
if __name__ == '__main__':
sys.exit(main())