Files
bdeshiandClaude Opus 5 82849eb94d Initial commit: Ross 1988 re-typeset edition
A searchable, re-typeset edition of Fiona Ross, "The Evolution of the
Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988),
transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178
plates and 410 inline type specimens cut from the scan; 51 errata.

Tracked: the transcription (src/pages), the preamble and its typographic
decisions, the cut images (plates/ — not reliably regenerable, the crop
specs for the inline cuts were never scripted), tools, and the four
working documents.

Not tracked: the built PDF, which `make` remakes from src/ and plates/;
the ProQuest scan under source/, which is third-party and needed only by
`make prep` and `make plate`; scans/ and work/, both regenerable.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 19:03:25 +06:00

77 lines
3.4 KiB
Python
Executable File

#!/usr/bin/env python3
"""Cut inline type-specimen glyphs out of a page at native resolution.
crop_inline.py PAGE NAME=SPEC [NAME=SPEC ...]
SPEC is an ink-run index from work/boxes<PAGE>.json, optionally with a side:
13 the whole run
60:right split the run at its widest internal gap, keep the right part
14:trimright drop only the rightmost element (a trailing comma, say)
6:mid drop the first and last elements, keeping the middle
5:last keep what follows the final gap (when the widest gap is elsewhere)
5+59 start 59 px into the run (last resort: no gap is detectable)
4~610 keep only the leftmost 610 px of the run
0-11 the span from run 0 to run 11 (for a whole specimen line)
Writes plates/inline/p<PAGE>-<NAME>.png (1-bit when the source is bitonal)."""
import subprocess, io, json, sys
import numpy as np
from PIL import Image
page = int(sys.argv[1])
boxes = json.load(open(f"work/boxes{page}.json"))
png = subprocess.run(["pdftoppm","-png","-r","300","-f",str(page),"-l",str(page),
"source/10731406.pdf"], capture_output=True).stdout
im = Image.open(io.BytesIO(png)).convert("L")
arr = np.array(im) < 128
def split(box, side):
x0, y0, x1, y1 = box
col = arr[y0:y1, x0:x1].mean(0) > 0.004
gaps, i = [], 0
while i < len(col):
if not col[i]:
j = i
while j < len(col) and not col[j]: j += 1
gaps.append((j-i, i, j)); i = j
else: i += 1
gaps = [g for g in gaps if g[1] > 0 and g[2] < len(col)]
if not gaps: return box
if side in ("left", "right"):
_, gi, gj = max(gaps)
return (x0, y0, x0+gi, y1) if side == "left" else (x0+gj, y0, x1, y1)
if side == "last": # keep only what follows the FINAL gap
return (x0+gaps[-1][2], y0, x1, y1)
if side == "first": # keep only what precedes the FIRST gap
return (x0, y0, x0+gaps[0][1], y1)
if side == "mid": # drop the first and last elements (`, G ,')
return (x0+gaps[0][2], y0, x0+gaps[-1][1], y1)
if side == "trimright": # drop the last element only
_, gi, gj = gaps[-1]
return (x0, y0, x0+gi, y1)
_, gi, gj = gaps[0] # trimleft: drop the first element only
return (x0+gj, y0, x1, y1)
for spec in sys.argv[2:]:
name, s = spec.split("=")
side = None
if ":" in s: s, side = s.split(":")
off = 0; keep = None
if "~" in s: # 4~610 — keep only the leftmost 610 px of run 4
s, k = s.split("~")
keep = int(k)
if "+" in s: # 5+59 — start 59 px into run 5, when the space
s, o = s.split("+") # between a word and the glyph is too noisy to
off = int(o) # register as a gap
if "-" in s:
a, b = (int(v) for v in s.split("-"))
box = (min(boxes[i][1] for i in range(a,b+1)), min(boxes[i][2] for i in range(a,b+1)),
max(boxes[i][3] for i in range(a,b+1)), max(boxes[i][4] for i in range(a,b+1)))
else:
box = tuple(boxes[int(s)][1:])
if side: box = split(box, side)
if off: box = (box[0]+off, box[1], box[2], box[3])
if keep: box = (box[0], box[1], box[0]+keep, box[3])
c = im.crop(box)
if len(np.unique(np.array(c))) <= 2: c = c.point(lambda v: 255 if v > 128 else 0).convert("1")
out = f"plates/inline/p{page:04d}-{name}.png"
c.save(out, optimize=True); print(out, c.size)