The scan is a bilevel 300 dpi JBIG2 layer plus lower-resolution colour, so storing every plate as 8-bit PNG only added size. tools/optimize_plates.py writes copies under work/plates-opt (bilevel to generic-mode, lossless JBIG2; genuine halftone to JPEG q88) and repoints work/pages at them; plates/ and src/ are untouched. The Dockerfile builds jbig2enc and adds qpdf, ghostscript and img2pdf. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
193 lines
7.6 KiB
Python
193 lines
7.6 KiB
Python
#!/usr/bin/env python3
|
|
r"""Re-encode the plate images with codecs matched to what they actually hold.
|
|
|
|
WHY
|
|
|
|
The ProQuest scan stores each page as MRC: a 300 dpi bilevel JBIG2 text layer
|
|
plus a 150/75 dpi JPX wash for paper tint. Our plates are cropped from a 300 dpi
|
|
re-rasterisation of that composite and saved as 8-bit greyscale PNG, which
|
|
stores, per pixel, about 60x what the source spends -- almost all of it
|
|
antialiasing our own render introduced and a background tint upsampled from
|
|
150 dpi. Measured on this scan:
|
|
|
|
source JBIG2 0.019 bits/pixel
|
|
CCITT G4 0.084
|
|
1-bit PNG 0.159
|
|
our 8-bit PNG ~1.2
|
|
|
|
The originals in plates/ are never modified. This writes re-encoded copies to
|
|
work/plates-opt/ and repoints work/pages/*.tex at them, so the transcription in
|
|
src/ is untouched and the whole thing reverts by deleting work/.
|
|
|
|
Safe because \plateop and \plate scale plates to fit the measure
|
|
(width=\hsize, keepaspectratio) -- pixel dimensions do not affect layout. The
|
|
inline specimens are NOT touched: \ig sets them at natural size (1px = 1/300in),
|
|
so their resolution is load-bearing, and they are only 1.6 MB in total.
|
|
|
|
CLASSIFICATION
|
|
|
|
A plate whose midtone fraction is low is type on paper: its real content is the
|
|
bilevel layer, and thresholding discards only our own resampling. A plate with
|
|
substantial midtone is a photograph or halftone, where grey IS the content;
|
|
those are kept 8-bit and JPEG-encoded instead.
|
|
"""
|
|
import argparse
|
|
import glob
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
|
|
import numpy as np
|
|
from PIL import Image
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
OUTDIR = os.path.join(ROOT, 'work', 'plates-opt')
|
|
|
|
|
|
def midtone(a):
|
|
h = np.bincount(a.ravel(), minlength=256)
|
|
return 1.0 - (h[:40].sum() + h[216:].sum()) / a.size
|
|
|
|
|
|
def _pdf_escape(n):
|
|
return str(n).encode()
|
|
|
|
|
|
def jbig2_pdf(png_path, out_pdf):
|
|
"""Encode bilevel with jbig2enc in GENERIC (lossless) mode and wrap the
|
|
stream in a one-page PDF.
|
|
|
|
Generic mode, NOT symbol mode. jbig2enc's `-s` builds a dictionary of
|
|
glyph shapes and substitutes visually similar symbols for one another --
|
|
it is lossy, and it is the mechanism behind the Xerox scanning affair
|
|
where scanned digits were silently swapped. Measured here it altered
|
|
1-4% of pixels. In a thesis whose argument IS the shape of the
|
|
letterforms, a substituted glyph would be a fabricated reading, so the
|
|
saving is not available to us at any price. Generic mode turned out to be
|
|
both lossless and slightly SMALLER on these single-page plates (19,749
|
|
bytes against 20,831 on p0063), because a per-page symbol dictionary does
|
|
not repay its own overhead.
|
|
|
|
Returns False if the encoder is unavailable, so the caller can fall back.
|
|
"""
|
|
if shutil.which('jbig2') is None:
|
|
return False
|
|
with tempfile.TemporaryDirectory() as td:
|
|
# -p emits the PDF-EMBEDDED stream form. Without it jbig2enc writes
|
|
# the standalone JBIG2 *file* format, header and all, which a PDF
|
|
# reader cannot decode -- it silently yields a blank page rather than
|
|
# an error, so this only shows up if you check the decoded pixels.
|
|
r = subprocess.run(['jbig2', '-p', os.path.abspath(png_path)],
|
|
cwd=td, capture_output=True)
|
|
if r.returncode != 0 or not r.stdout:
|
|
return False
|
|
image_data = r.stdout
|
|
im = Image.open(png_path)
|
|
w, h = im.size
|
|
pw, ph = w * 72.0 / 300.0, h * 72.0 / 300.0
|
|
|
|
objs = []
|
|
objs.append(b"<< /Type /Catalog /Pages 2 0 R >>")
|
|
objs.append(b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>")
|
|
objs.append(("<< /Type /Page /Parent 2 0 R /MediaBox [0 0 %.2f %.2f] "
|
|
"/Resources << /XObject << /Im0 4 0 R >> >> /Contents 5 0 R >>"
|
|
% (pw, ph)).encode())
|
|
objs.append(("<< /Type /XObject /Subtype /Image /Width %d /Height %d "
|
|
"/ColorSpace /DeviceGray /BitsPerComponent 1 "
|
|
# no /Decode: JBIG2Decode already yields PDF's polarity
|
|
# (0 = black). Adding /Decode [1 0] inverts the plate, which
|
|
# showed up as a ~90% pixel mismatch against the original.
|
|
"/Filter /JBIG2Decode /Length %d >>\nstream\n"
|
|
% (w, h, len(image_data))).encode()
|
|
+ image_data + b"\nendstream")
|
|
content = ("q %.2f 0 0 %.2f 0 0 cm /Im0 Do Q" % (pw, ph)).encode()
|
|
objs.append(b"<< /Length " + _pdf_escape(len(content)) + b" >>\nstream\n"
|
|
+ content + b"\nendstream")
|
|
|
|
out = bytearray(b"%PDF-1.4\n")
|
|
offsets = []
|
|
for i, body in enumerate(objs, start=1):
|
|
offsets.append(len(out))
|
|
out += b"%d 0 obj\n" % i + body + b"\nendobj\n"
|
|
xref = len(out)
|
|
out += b"xref\n0 %d\n" % (len(objs) + 1)
|
|
out += b"0000000000 65535 f \n"
|
|
for off in offsets:
|
|
out += b"%010d 00000 n \n" % off
|
|
out += (b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n"
|
|
% (len(objs) + 1, xref))
|
|
with open(out_pdf, 'wb') as f:
|
|
f.write(bytes(out))
|
|
return True
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument('--tone-quality', type=int, default=88)
|
|
ap.add_argument('--tone-threshold', type=float, default=0.25,
|
|
help='midtone fraction above which a plate keeps 8-bit grey')
|
|
ap.add_argument('--rewrite', default='work/pages',
|
|
help='rewrite \\includegraphics paths in these .tex files')
|
|
ap.add_argument('--dry-run', action='store_true')
|
|
args = ap.parse_args()
|
|
|
|
os.makedirs(OUTDIR, exist_ok=True)
|
|
mapping = {}
|
|
n_b = n_t = 0
|
|
before = after = 0
|
|
used_jbig2 = used_g4 = 0
|
|
|
|
for src in sorted(glob.glob(os.path.join(ROOT, 'plates', 'p0*.png'))):
|
|
if not re.search(r'p\d+\.png$', src):
|
|
continue
|
|
rel = os.path.relpath(src, ROOT)
|
|
a = np.array(Image.open(src).convert('L'))
|
|
sz = os.path.getsize(src)
|
|
before += sz
|
|
stem = os.path.splitext(os.path.basename(src))[0]
|
|
if midtone(a) > args.tone_threshold:
|
|
dst = os.path.join(OUTDIR, stem + '.jpg')
|
|
if not args.dry_run:
|
|
Image.fromarray(a).save(dst, 'JPEG', quality=args.tone_quality,
|
|
optimize=True)
|
|
n_t += 1
|
|
else:
|
|
dst = os.path.join(OUTDIR, stem + '.pdf')
|
|
if not args.dry_run:
|
|
bw = os.path.join(OUTDIR, stem + '.bw.png')
|
|
Image.fromarray((a > 128)).save(bw, 'PNG', bits=1)
|
|
if jbig2_pdf(bw, dst):
|
|
used_jbig2 += 1
|
|
else:
|
|
Image.fromarray((a > 128)).save(dst, 'PDF') # CCITT G4
|
|
used_g4 += 1
|
|
os.remove(bw)
|
|
n_b += 1
|
|
if not args.dry_run:
|
|
after += os.path.getsize(dst)
|
|
mapping[rel] = os.path.relpath(dst, ROOT)
|
|
|
|
if not args.dry_run and args.rewrite:
|
|
pat = re.compile(r'plates/(p\d+)\.png')
|
|
changed = 0
|
|
for tex in glob.glob(os.path.join(ROOT, args.rewrite, '*.tex')):
|
|
s = open(tex).read()
|
|
new = pat.sub(lambda m: mapping.get(f'plates/{m.group(1)}.png',
|
|
m.group(0)), s)
|
|
if new != s:
|
|
open(tex, 'w').write(new)
|
|
changed += 1
|
|
print(f'rewrote {changed} page files to point at work/plates-opt/')
|
|
|
|
print(f'bilevel {n_b} (jbig2 {used_jbig2}, g4 fallback {used_g4}), tonal {n_t}')
|
|
if after:
|
|
print(f'plates: {before/1048576:.1f} MB -> {after/1048576:.1f} MB '
|
|
f'({after/before:.1%})')
|
|
|
|
|
|
if __name__ == '__main__':
|
|
sys.exit(main())
|