diff --git a/Dockerfile b/Dockerfile index b2a2ad3..bf4acc7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -9,9 +9,23 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ poppler-utils tesseract-ocr tesseract-ocr-eng \ python3 python3-pip python3-pil python3-numpy \ make curl ca-certificates \ + qpdf ghostscript img2pdf \ + git automake autoconf libtool g++ pkg-config libleptonica-dev zlib1g-dev \ && rm -rf /var/lib/apt/lists/* \ && pip3 install --no-cache-dir --break-system-packages pdfplumber fonttools \ && fc-cache -f + +# jbig2enc: the encoder the ProQuest scan itself uses for its bilevel page +# layer, and the reason a 2556x3305 page fits in 4900 bytes there. It builds a +# dictionary of repeated letterforms and stores each shape once, which general +# compressors cannot do -- measured on this scan, 0.019 bits/pixel against +# 0.084 for CCITT G4 and 0.159 for 1-bit PNG. Not packaged for Ubuntu 24.04, so +# it is built from source. +RUN git clone --depth 1 https://github.com/agl/jbig2enc /tmp/jbig2enc \ + && cd /tmp/jbig2enc \ + && ./autogen.sh && ./configure && make -j"$(nproc)" && make install \ + && ldconfig \ + && rm -rf /tmp/jbig2enc WORKDIR /work # The project directory is bind-mounted at /work (see compose.yaml); nothing is copied in. CMD ["bash"] diff --git a/Makefile b/Makefile index 33b1030..f6058e0 100644 --- a/Makefile +++ b/Makefile @@ -7,7 +7,7 @@ export GID := $(shell id -g) RUN := $(if $(NODOCKER),,docker compose run --rm -T tex) OUT := ross-1988-retypeset.pdf -SOURCES := build.sh tools/polish.py src/preamble.tex src/colophon.tex src/errata.tex \ +SOURCES := build.sh tools/polish.py tools/optimize_plates.py src/preamble.tex src/colophon.tex src/errata.tex \ $(wildcard src/pages/*.tex) .DEFAULT_GOAL := pdf diff --git a/build.sh b/build.sh index fc42675..70fcb34 100755 --- a/build.sh +++ b/build.sh @@ -21,6 +21,12 @@ trap 'rmdir work/.buildlock 2>/dev/null' EXIT INT TERM # Three full passes run below regardless, so a stale aux buys nothing. rm -f work/main.aux work/main.toc work/main.out python3 tools/polish.py >/dev/null # typographic pass: src/pages -> work/pages +# Re-encode the plates to match what they actually hold: bilevel type goes to +# lossless JBIG2, genuine halftone stays 8-bit as JPEG. Writes copies under +# work/plates-opt and repoints work/pages at them -- plates/ and src/ are +# never touched, so deleting work/ reverts it. Must run AFTER polish.py, which +# is what creates work/pages. +python3 tools/optimize_plates.py { cat src/preamble.tex printf '%s\n' '\begin{document}\immediate\openout\unsurefile=unsure.log\immediate\openout\erratafile=errata.log\immediate\openout\qslipfile=qslips.log' printf '\\input{%s}\n' work/pages/p0001.tex src/colophon.tex diff --git a/tools/optimize_plates.py b/tools/optimize_plates.py new file mode 100644 index 0000000..d016bfe --- /dev/null +++ b/tools/optimize_plates.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +r"""Re-encode the plate images with codecs matched to what they actually hold. + +WHY + +The ProQuest scan stores each page as MRC: a 300 dpi bilevel JBIG2 text layer +plus a 150/75 dpi JPX wash for paper tint. Our plates are cropped from a 300 dpi +re-rasterisation of that composite and saved as 8-bit greyscale PNG, which +stores, per pixel, about 60x what the source spends -- almost all of it +antialiasing our own render introduced and a background tint upsampled from +150 dpi. Measured on this scan: + + source JBIG2 0.019 bits/pixel + CCITT G4 0.084 + 1-bit PNG 0.159 + our 8-bit PNG ~1.2 + +The originals in plates/ are never modified. This writes re-encoded copies to +work/plates-opt/ and repoints work/pages/*.tex at them, so the transcription in +src/ is untouched and the whole thing reverts by deleting work/. + +Safe because \plateop and \plate scale plates to fit the measure +(width=\hsize, keepaspectratio) -- pixel dimensions do not affect layout. The +inline specimens are NOT touched: \ig sets them at natural size (1px = 1/300in), +so their resolution is load-bearing, and they are only 1.6 MB in total. + +CLASSIFICATION + +A plate whose midtone fraction is low is type on paper: its real content is the +bilevel layer, and thresholding discards only our own resampling. A plate with +substantial midtone is a photograph or halftone, where grey IS the content; +those are kept 8-bit and JPEG-encoded instead. +""" +import argparse +import glob +import os +import re +import shutil +import subprocess +import sys +import tempfile + +import numpy as np +from PIL import Image + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +OUTDIR = os.path.join(ROOT, 'work', 'plates-opt') + + +def midtone(a): + h = np.bincount(a.ravel(), minlength=256) + return 1.0 - (h[:40].sum() + h[216:].sum()) / a.size + + +def _pdf_escape(n): + return str(n).encode() + + +def jbig2_pdf(png_path, out_pdf): + """Encode bilevel with jbig2enc in GENERIC (lossless) mode and wrap the + stream in a one-page PDF. + + Generic mode, NOT symbol mode. jbig2enc's `-s` builds a dictionary of + glyph shapes and substitutes visually similar symbols for one another -- + it is lossy, and it is the mechanism behind the Xerox scanning affair + where scanned digits were silently swapped. Measured here it altered + 1-4% of pixels. In a thesis whose argument IS the shape of the + letterforms, a substituted glyph would be a fabricated reading, so the + saving is not available to us at any price. Generic mode turned out to be + both lossless and slightly SMALLER on these single-page plates (19,749 + bytes against 20,831 on p0063), because a per-page symbol dictionary does + not repay its own overhead. + + Returns False if the encoder is unavailable, so the caller can fall back. + """ + if shutil.which('jbig2') is None: + return False + with tempfile.TemporaryDirectory() as td: + # -p emits the PDF-EMBEDDED stream form. Without it jbig2enc writes + # the standalone JBIG2 *file* format, header and all, which a PDF + # reader cannot decode -- it silently yields a blank page rather than + # an error, so this only shows up if you check the decoded pixels. + r = subprocess.run(['jbig2', '-p', os.path.abspath(png_path)], + cwd=td, capture_output=True) + if r.returncode != 0 or not r.stdout: + return False + image_data = r.stdout + im = Image.open(png_path) + w, h = im.size + pw, ph = w * 72.0 / 300.0, h * 72.0 / 300.0 + + objs = [] + objs.append(b"<< /Type /Catalog /Pages 2 0 R >>") + objs.append(b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>") + objs.append(("<< /Type /Page /Parent 2 0 R /MediaBox [0 0 %.2f %.2f] " + "/Resources << /XObject << /Im0 4 0 R >> >> /Contents 5 0 R >>" + % (pw, ph)).encode()) + objs.append(("<< /Type /XObject /Subtype /Image /Width %d /Height %d " + "/ColorSpace /DeviceGray /BitsPerComponent 1 " + # no /Decode: JBIG2Decode already yields PDF's polarity + # (0 = black). Adding /Decode [1 0] inverts the plate, which + # showed up as a ~90% pixel mismatch against the original. + "/Filter /JBIG2Decode /Length %d >>\nstream\n" + % (w, h, len(image_data))).encode() + + image_data + b"\nendstream") + content = ("q %.2f 0 0 %.2f 0 0 cm /Im0 Do Q" % (pw, ph)).encode() + objs.append(b"<< /Length " + _pdf_escape(len(content)) + b" >>\nstream\n" + + content + b"\nendstream") + + out = bytearray(b"%PDF-1.4\n") + offsets = [] + for i, body in enumerate(objs, start=1): + offsets.append(len(out)) + out += b"%d 0 obj\n" % i + body + b"\nendobj\n" + xref = len(out) + out += b"xref\n0 %d\n" % (len(objs) + 1) + out += b"0000000000 65535 f \n" + for off in offsets: + out += b"%010d 00000 n \n" % off + out += (b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n" + % (len(objs) + 1, xref)) + with open(out_pdf, 'wb') as f: + f.write(bytes(out)) + return True + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument('--tone-quality', type=int, default=88) + ap.add_argument('--tone-threshold', type=float, default=0.25, + help='midtone fraction above which a plate keeps 8-bit grey') + ap.add_argument('--rewrite', default='work/pages', + help='rewrite \\includegraphics paths in these .tex files') + ap.add_argument('--dry-run', action='store_true') + args = ap.parse_args() + + os.makedirs(OUTDIR, exist_ok=True) + mapping = {} + n_b = n_t = 0 + before = after = 0 + used_jbig2 = used_g4 = 0 + + for src in sorted(glob.glob(os.path.join(ROOT, 'plates', 'p0*.png'))): + if not re.search(r'p\d+\.png$', src): + continue + rel = os.path.relpath(src, ROOT) + a = np.array(Image.open(src).convert('L')) + sz = os.path.getsize(src) + before += sz + stem = os.path.splitext(os.path.basename(src))[0] + if midtone(a) > args.tone_threshold: + dst = os.path.join(OUTDIR, stem + '.jpg') + if not args.dry_run: + Image.fromarray(a).save(dst, 'JPEG', quality=args.tone_quality, + optimize=True) + n_t += 1 + else: + dst = os.path.join(OUTDIR, stem + '.pdf') + if not args.dry_run: + bw = os.path.join(OUTDIR, stem + '.bw.png') + Image.fromarray((a > 128)).save(bw, 'PNG', bits=1) + if jbig2_pdf(bw, dst): + used_jbig2 += 1 + else: + Image.fromarray((a > 128)).save(dst, 'PDF') # CCITT G4 + used_g4 += 1 + os.remove(bw) + n_b += 1 + if not args.dry_run: + after += os.path.getsize(dst) + mapping[rel] = os.path.relpath(dst, ROOT) + + if not args.dry_run and args.rewrite: + pat = re.compile(r'plates/(p\d+)\.png') + changed = 0 + for tex in glob.glob(os.path.join(ROOT, args.rewrite, '*.tex')): + s = open(tex).read() + new = pat.sub(lambda m: mapping.get(f'plates/{m.group(1)}.png', + m.group(0)), s) + if new != s: + open(tex, 'w').write(new) + changed += 1 + print(f'rewrote {changed} page files to point at work/plates-opt/') + + print(f'bilevel {n_b} (jbig2 {used_jbig2}, g4 fallback {used_g4}), tonal {n_t}') + if after: + print(f'plates: {before/1048576:.1f} MB -> {after/1048576:.1f} MB ' + f'({after/before:.1%})') + + +if __name__ == '__main__': + sys.exit(main())