Initial commit: Ross 1988 re-typeset edition

A searchable, re-typeset edition of Fiona Ross, "The Evolution of the
Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988),
transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178
plates and 410 inline type specimens cut from the scan; 51 errata.

Tracked: the transcription (src/pages), the preamble and its typographic
decisions, the cut images (plates/ — not reliably regenerable, the crop
specs for the inline cuts were never scripted), tools, and the four
working documents.

Not tracked: the built PDF, which `make` remakes from src/ and plates/;
the ProQuest scan under source/, which is third-party and needed only by
`make prep` and `make plate`; scans/ and work/, both regenerable.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-14 19:03:25 +06:00
co-authored by Claude Opus 5
commit 82849eb94d
1053 changed files with 6089 additions and 0 deletions
Binary file not shown.
Executable
+54
View File
@@ -0,0 +1,54 @@
#!/usr/bin/env python3
r"""Ratchet checks for the re-typeset project. Exit 1 on any hard failure.
- every manifest row marked done points at an existing page file (or is dropped/colophon)
- every src/pages file is a manifest row marked done
- page file carries \origpage{P}/\plateop{P} with P == manifest.printed
- a head macro (\chaphead/\chapstart/\chapnum/\partstart/\sectionstart) precedes \origpage
- printed page numbers increase in PDF order; note numbers \fn{n} run 1,2,3… per chapter
- lists \unsure{} readings so they can be mirrored in QUESTIONS.md
- every \erratum{corrected}{as printed} has a matching \erratumline in src/errata.tex
"""
import csv, re, glob, os, sys
fail=[]; warn=[]
rows=list(csv.DictReader(open("manifest.tsv"),delimiter="\t"))
files=set(glob.glob("src/pages/p*.tex"))
seen=set(); last_printed=0; fn_expect=1; unsure=[]; errata_used=[]
for r in sorted(rows,key=lambda r:int(r["pdf_page"])):
p=int(r["pdf_page"]); st=r["status"]; kind=r["kind"]; f=r["src_file"]
if st!="done": continue
if kind in ("dropped","colophon"): continue
if not f or not os.path.exists(f): fail.append(f"PDF {p}: done but src_file missing ({f!r})"); continue
seen.add(f); s=open(f).read()
m=re.search(r"\\(?:origpage|plateop)\{(\d+)\}",s)
if not m: fail.append(f"{f}: no \\origpage/\\plateop")
else:
pr=int(m.group(1))
if r["printed"] and pr!=int(r["printed"]): fail.append(f"{f}: \\origpage{{{pr}}} but manifest printed={r['printed']}")
if pr<=last_printed: fail.append(f"{f}: printed {pr} not after previous {last_printed}")
last_printed=pr
heads = ("\\chaphead", "\\chapstart", "\\partstart", "\\sectionstart", "\\chapnum")
if any(h in s for h in heads):
if s.find("\\origpage")!=-1 and s.find("\\origpage")<max(s.find(h) for h in heads):
fail.append(f"{f}: \\origpage appears before \\chaphead/\\chapstart (would leave an empty page)")
fn_expect=1
for n in re.findall(r"\\fn\{(\d+)\}",s):
if int(n)!=fn_expect: fail.append(f"{f}: \\fn{{{n}}} but expected {fn_expect}")
fn_expect=int(n)+1
unsure+= [f"{f}: {u}" for u in re.findall(r"\\unsure\{([^}]*)\}",s)]
# \erratum args must be plain text (no nested braces): put \emph outside it
errata_used += [(r["printed"], corr, orig, f)
for corr, orig in re.findall(r"\\erratum\{([^}]*)\}\{([^}]*)\}", s)]
errata_src = open("src/errata.tex").read() if os.path.exists("src/errata.tex") else ""
listed = set(re.findall(r"\\erratumline\{(\d+)\}\{([^}]*)\}\{([^}]*)\}", errata_src))
for pg, corr, orig, f in errata_used:
if (pg, orig, corr) not in listed:
fail.append(f"{f}: \\erratum{{{corr}}}{{{orig}}} has no \\erratumline{{{pg}}}{{{orig}}}{{{corr}}} in src/errata.tex")
for pg, orig, corr in sorted(listed):
if not any(pg==u[0] and orig==u[2] and corr==u[1] for u in errata_used):
fail.append(f"src/errata.tex: \\erratumline{{{pg}}}{{{orig}}}{{{corr}}} has no \\erratum in a page file")
for f in sorted(files-seen): fail.append(f"{f} exists but manifest row is not done/linked")
done=sum(1 for r in rows if r["status"]=="done"); print(f"progress: {done}/{len(rows)} PDF pages done")
if unsure: print("unsure readings (mirror in QUESTIONS.md):"); [print(" ",u) for u in unsure]
for w in warn: print("WARN",w)
for e in fail: print("FAIL",e)
sys.exit(1 if fail else 0)
+76
View File
@@ -0,0 +1,76 @@
#!/usr/bin/env python3
"""Cut inline type-specimen glyphs out of a page at native resolution.
crop_inline.py PAGE NAME=SPEC [NAME=SPEC ...]
SPEC is an ink-run index from work/boxes<PAGE>.json, optionally with a side:
13 the whole run
60:right split the run at its widest internal gap, keep the right part
14:trimright drop only the rightmost element (a trailing comma, say)
6:mid drop the first and last elements, keeping the middle
5:last keep what follows the final gap (when the widest gap is elsewhere)
5+59 start 59 px into the run (last resort: no gap is detectable)
4~610 keep only the leftmost 610 px of the run
0-11 the span from run 0 to run 11 (for a whole specimen line)
Writes plates/inline/p<PAGE>-<NAME>.png (1-bit when the source is bitonal)."""
import subprocess, io, json, sys
import numpy as np
from PIL import Image
page = int(sys.argv[1])
boxes = json.load(open(f"work/boxes{page}.json"))
png = subprocess.run(["pdftoppm","-png","-r","300","-f",str(page),"-l",str(page),
"source/10731406.pdf"], capture_output=True).stdout
im = Image.open(io.BytesIO(png)).convert("L")
arr = np.array(im) < 128
def split(box, side):
x0, y0, x1, y1 = box
col = arr[y0:y1, x0:x1].mean(0) > 0.004
gaps, i = [], 0
while i < len(col):
if not col[i]:
j = i
while j < len(col) and not col[j]: j += 1
gaps.append((j-i, i, j)); i = j
else: i += 1
gaps = [g for g in gaps if g[1] > 0 and g[2] < len(col)]
if not gaps: return box
if side in ("left", "right"):
_, gi, gj = max(gaps)
return (x0, y0, x0+gi, y1) if side == "left" else (x0+gj, y0, x1, y1)
if side == "last": # keep only what follows the FINAL gap
return (x0+gaps[-1][2], y0, x1, y1)
if side == "first": # keep only what precedes the FIRST gap
return (x0, y0, x0+gaps[0][1], y1)
if side == "mid": # drop the first and last elements (`, G ,')
return (x0+gaps[0][2], y0, x0+gaps[-1][1], y1)
if side == "trimright": # drop the last element only
_, gi, gj = gaps[-1]
return (x0, y0, x0+gi, y1)
_, gi, gj = gaps[0] # trimleft: drop the first element only
return (x0+gj, y0, x1, y1)
for spec in sys.argv[2:]:
name, s = spec.split("=")
side = None
if ":" in s: s, side = s.split(":")
off = 0; keep = None
if "~" in s: # 4~610 — keep only the leftmost 610 px of run 4
s, k = s.split("~")
keep = int(k)
if "+" in s: # 5+59 — start 59 px into run 5, when the space
s, o = s.split("+") # between a word and the glyph is too noisy to
off = int(o) # register as a gap
if "-" in s:
a, b = (int(v) for v in s.split("-"))
box = (min(boxes[i][1] for i in range(a,b+1)), min(boxes[i][2] for i in range(a,b+1)),
max(boxes[i][3] for i in range(a,b+1)), max(boxes[i][4] for i in range(a,b+1)))
else:
box = tuple(boxes[int(s)][1:])
if side: box = split(box, side)
if off: box = (box[0]+off, box[1], box[2], box[3])
if keep: box = (box[0], box[1], box[0]+keep, box[3])
c = im.crop(box)
if len(np.unique(np.array(c))) <= 2: c = c.point(lambda v: 255 if v > 128 else 0).convert("1")
out = f"plates/inline/p{page:04d}-{name}.png"
c.save(out, optimize=True); print(out, c.size)
+225
View File
@@ -0,0 +1,225 @@
#!/usr/bin/env python3
"""Extract a plate image from the original scan, dropping the printed page
number (top) and the typeset caption (bottom), which are re-set in LaTeX.
usage: crop_plate.py PDF_PAGE [--top FRAC --bottom FRAC] [--keep-caption]
Writes plates/pNNNN.png (1-bit PNG when the source is bitonal).
Heuristic defaults: ignore the outer 4% (scanner edge), the top 7.5%
(page number) and the bottom 9% (caption). Override per page via manifest.
"""
import sys, argparse, subprocess, io, math
from PIL import Image, ImageOps
import numpy as np
import os
SRC = os.environ.get("ROSS_SRC", "source/10731406.pdf")
ap = argparse.ArgumentParser()
ap.add_argument("page", type=int)
ap.add_argument("--top", type=float, default=0.075)
ap.add_argument("--bottom", type=float, default=0.09)
ap.add_argument("--edge", type=float, default=0.04)
ap.add_argument("--dpi", type=int, default=300)
ap.add_argument("--box", default=None,
help="explicit crop as LEFT,TOP,RIGHT,BOTTOM fractions of the raw page; skips the scanner-bar and page-number heuristics. Use for dark plates (a photographed manuscript), where \"dark = scanner bar\" does not hold")
ap.add_argument("--trim", default=None,
help="post-rotation trim as TOP,RIGHT,BOTTOM,LEFT fractions, e.g. 0,0,0.06,0")
ap.add_argument("--rotate", type=int, default=0, choices=[0,90,180,270],
help="rotate the crop counter-clockwise (lossless); plates the original prints sideways")
ap.add_argument("--raw", action="store_true",
help="with --box: take the box exactly — no shadow sweep, deskew, bar sweep or autocrop. Use for a framed plate, whose own border rules the edge heuristics mistake for scanner bars")
ap.add_argument("--rule", type=float, default=0.012,
help="dark runs thinner than this fraction are plate rules, not scanner bars")
a = ap.parse_args()
png = subprocess.run(["pdftoppm", "-png", "-r", str(a.dpi), "-f", str(a.page),
"-l", str(a.page), SRC], capture_output=True).stdout
im = Image.open(io.BytesIO(png)).convert("L")
W, H = im.size
# strip scanner black bars: any column/row that is >40% dark is an edge;
# keep only the widest run of "paper" columns/rows.
dark = np.array(im) < 100
def paper_run(profile, rule=None):
ok = profile < 0.40
# A thin dark run is a ruled line belonging to the plate (engraved column
# rules, table borders), not a scanner bar: close it so it cannot split the
# paper run. Scanner bars are an order of magnitude thicker.
rule = int((rule if rule is not None else a.rule) * len(ok))
i = 0
while i < len(ok):
if not ok[i]:
j = i
while j < len(ok) and not ok[j]: j += 1
interior = i > 0.05*len(ok) and j < 0.95*len(ok)
if j - i <= rule and interior: ok[i:j] = True
i = j
else: i += 1
best=(0,0); i=0
while i < len(ok):
if ok[i]:
j=i
while j < len(ok) and ok[j]: j+=1
if j-i > best[1]-best[0]: best=(i,j)
i=j
else: i+=1
return best
if a.box:
l, t, r, b = (float(x) for x in a.box.split(","))
im = im.crop((int(W*l), int(H*t), int(W*r), int(H*b)))
else:
c0,c1 = paper_run(dark.mean(0)); r0,r1 = paper_run(dark.mean(1))
im = im.crop((c0,r0,c1,r1)); W,H = im.size
box = (int(W*a.edge), int(H*a.top), int(W*(1-a.edge)), int(H*(1-a.bottom)))
im = im.crop(box)
# A thin, very dark run just inside an edge is a scanner bar the paper-run
# heuristic kept (it is thinner than --rule, so it was read as a plate rule).
# Sweep it off before autocropping.
def strip_edge_bars(im, frac=0.06, maxthick=30):
for _ in range(2):
a = np.array(im) < 100
W, H = im.size
l, r, t, b = 0, W, 0, H
cols = a.mean(0); rows = a.mean(1)
for i in range(int(W*frac)):
if cols[i] > 0.5: l = i + 1
for i in range(W - 1, W - int(W*frac) - 1, -1):
if cols[i] > 0.5: r = i
for i in range(int(H*frac)):
if rows[i] > 0.5: t = i + 1
for i in range(H - 1, H - int(H*frac) - 1, -1):
if rows[i] > 0.5: b = i
if (l, r, t, b) == (0, W, 0, H): break
if l > maxthick + int(W*frac) or W - r > maxthick + int(W*frac): break
im = im.crop((l, t, r, b))
return im
# ---- deskew -----------------------------------------------------------
# Scans are a degree or two out of square: the page edge and any ruled lines
# lean. Estimate the angle by rotating a downsampled copy through a small range
# and taking the angle whose row/column ink profiles are sharpest (a straight
# page concentrates ink into rows and columns, maximising their variance).
# NB rotation resamples — the only step in this tool that does. It is applied
# only when the page is measurably out of square.
def skew_angle(im, limit=2.0, step=0.05):
# measure on the interior: edge bands and the page frame would otherwise
# dominate the profile variance and pin the estimate at zero
W, H = im.size
g = im.crop((int(W*0.08), int(H*0.06), int(W*0.92), int(H*0.94)))
g = g.resize((max(g.width//4, 1), max(g.height//4, 1)), Image.BILINEAR)
best, best_score = 0.0, -1.0
n = int(limit/step)
for i in range(-n, n+1):
deg = round(i*step, 2)
a = np.array(g.rotate(deg, resample=Image.BILINEAR, fillcolor=255)) < 128
score = float(a.mean(1).var() + a.mean(0).var())
if score > best_score: best, best_score = deg, score
return best
def rule_angle(im):
"""Angle from long near-vertical or near-horizontal rules, when the plate
has any: compare where a rule sits near one end against the other. More
sensitive than the profile method for engravings, which have few text
lines but strong ruled columns."""
a = np.array(im) < 128
H, W = a.shape
def drift(arr, long_axis):
n = arr.shape[long_axis]
lo = arr.take(range(int(n*0.10), int(n*0.35)), axis=long_axis)
hi = arr.take(range(int(n*0.65), int(n*0.90)), axis=long_axis)
pl, ph = lo.mean(long_axis), hi.mean(long_axis)
# A rule is several pixels wide, so group contiguous strong lines and
# compare their CENTRES — matching column to column would pair edge with
# edge and under-measure the drift.
def centres(prof):
idx = [i for i in range(len(prof)) if prof[i] > 0.55]
if not idx: return []
out, cur = [], [idx[0]]
for i in idx[1:]:
if i - cur[-1] <= 3: cur.append(i)
else: out.append(sum(cur)/len(cur)); cur = [i]
out.append(sum(cur)/len(cur))
return out
cl, ch = centres(pl), centres(ph)
if not cl or not ch: return None
ds = []
for i in cl:
j = min(ch, key=lambda k: abs(k-i))
if abs(j-i) <= 20: ds.append(j-i)
if not ds: return None
span = n*0.55
return math.degrees(math.atan(float(np.median(ds))/span))
v = drift(a, 0) # vertical rules: drift measured down the page
h = drift(a.T, 0) # horizontal rules
cands = [x for x in (v, h) if x is not None]
if not cands: return None
return -cands[0] if abs(cands[0]) >= 0.05 else 0.0
def deskew(im):
ang = rule_angle(im)
if ang is None: ang = skew_angle(im)
if abs(ang) < 0.05: return im, 0.0
return im.rotate(ang, resample=Image.BICUBIC, expand=True, fillcolor=255), ang
# ---- scanner shadow ---------------------------------------------------------
# A soft grey band down an edge (the gutter shadow) is not dark enough for the
# bar sweep but still prints. Trim edge rows/columns that are mostly dark.
def strip_edge_shadow(im, frac=0.05, dark=0.30):
# A gutter shadow is often a wedge — dark over part of the edge only — so a
# whole-column mean misses it. Score each edge column (row) by the darkest
# tenth of its length instead.
a = np.array(im) < 150
W, H = im.size
l, r, t, b = 0, W, 0, H
def worst(v, n): # darkest window of length n along v
if len(v) < n: return v.mean() if len(v) else 0.0
c = np.cumsum(np.insert(v, 0, 0.0))
return float(((c[n:] - c[:-n])/n).max())
hw, vw = max(H//10, 1), max(W//10, 1)
# stop at the first light line: only a band touching the edge is shadow,
# anything past it is the plate's own content
for i in range(int(W*frac)):
if worst(a[:, i], hw) > dark: l = i + 1
else: break
for i in range(W-1, W-int(W*frac)-1, -1):
if worst(a[:, i], hw) > dark: r = i
else: break
for i in range(int(H*frac)):
if worst(a[i, :], vw) > dark: t = i + 1
else: break
for i in range(H-1, H-int(H*frac)-1, -1):
if worst(a[i, :], vw) > dark: b = i
else: break
return im.crop((l, t, r, b)) if (l, r, t, b) != (0, W, 0, H) else im
if a.raw:
_skew = 0.0
else:
im = strip_edge_shadow(im) # drop the gutter shadow first…
im, _skew = deskew(im) # …so it cannot bias the angle estimate
im = strip_edge_shadow(im) # …then clear what the rotation brought in
# autocrop to dark content with a small margin
def autocrop(im):
arr = np.array(im) < 128
rows = np.where(arr.mean(1) > 0.002)[0]; cols = np.where(arr.mean(0) > 0.002)[0]
if not (len(rows) and len(cols)): return im
m = int(0.01*im.width)
return im.crop((max(cols[0]-m,0), max(rows[0]-m,0),
min(cols[-1]+m, im.width), min(rows[-1]+m, im.height)))
if not a.raw:
im = autocrop(im) # bring the edges to the ink…
im = autocrop(strip_edge_bars(im)) # …then sweep off any scanner bar now at an edge
im = autocrop(strip_edge_shadow(im)) # …and the gutter shadow the autocrop just exposed
if a.rotate:
im = im.transpose({90: Image.ROTATE_90, 180: Image.ROTATE_180,
270: Image.ROTATE_270}[a.rotate])
# --trim T,R,B,L (fractions, after rotation): for skewed scanner edges that are
# too thin for the bar heuristic and too dark for the autocrop to ignore.
if a.trim:
t, r, b, l = (float(x) for x in a.trim.split(","))
W, H = im.size
im = autocrop(im.crop((int(W*l), int(H*t), int(W*(1-r)), int(H*(1-b)))))
uniq = np.unique(np.array(im))
out = f"plates/p{a.page:04d}.png"
if len(uniq) <= 2: # bitonal source: keep it crisp and small
im = im.point(lambda v: 255 if v > 128 else 0).convert("1")
im.save(out, optimize=True)
print(out, im.size, im.mode, f"deskew {_skew:+.1f}deg" if _skew else "")
+25
View File
@@ -0,0 +1,25 @@
#!/usr/bin/env python3
"""ocr2tex.py PDFPAGE PRINTED [--head 'Title'] — turn tesseract scaffold into a draft
page file (paragraphs joined, LaTeX specials escaped, quotes normalised).
The draft MUST then be proofread against the page image and corrected."""
import sys, re, argparse
ap=argparse.ArgumentParser(); ap.add_argument("page",type=int); ap.add_argument("printed",type=int)
ap.add_argument("--head",default=None); ap.add_argument("--bookmark",default=None)
a=ap.parse_args()
txt=open(f"work/ocr/p{a.page:04d}.txt").read()
lines=txt.splitlines()
if a.head and lines and lines[0].strip().lower()==a.head.lower(): lines=lines[1:]
body="\n".join(lines).strip()
paras=[re.sub(r"\s*\n\s*"," ",p).strip() for p in re.split(r"\n\s*\n",body) if p.strip()]
def esc(s):
s=s.replace("\\","\\textbackslash{}")
for c in "&%$#_{}": s=s.replace(c,"\\"+c)
s=re.sub(r"(\d)-(\d)",r"\1-\2",s)
s=s.replace("‘","`").replace("’","'").replace("“","``").replace("”","''")
s=re.sub(r"\b(\w+)-\s+(\w)",r"\1\2",s) # de-hyphenate line breaks (check!)
return s
out=[f"% PDF page {a.page} — printed page {a.printed}",f"\\origpage{{{a.printed}}}"]
if a.bookmark: out.append(f"\\pdfbookmark[1]{{{a.bookmark}}}{{bm{a.page}}}")
if a.head: out.append(f"\\chaphead{{{a.head}}}")
out.append("\n\n".join(esc(p) for p in paras))
p=f"src/pages/p{a.page:04d}.tex"; open(p,"w").write("\n".join(out)+"\n"); print(p, len(paras),"paras")
+164
View File
@@ -0,0 +1,164 @@
#!/usr/bin/env python3
r"""polish.py — build-time typographic pass over the transcription.
Reads src/pages/*.tex and writes work/pages/*.tex. The transcription itself stays
exactly as typed; this pass applies at set time the typography the author's
typewriter could not produce, plus the cross-reference links a digital edition
should have:
* ties, so a reference or a unit never breaks across a line
* en-dashes for numeric ranges (1778-1978, pp. 315-320)
* small caps for institutional acronyms (explicit whitelist below)
* p./pp., pl./pls. and chapter references become links to the original page
they name — note that the original's page numbers are the targets, not the
edition's, since `\origpage{N}` plants `page.N` where original page N begins
* \mbox round an inline specimen so it never breaks from adjacent punctuation
Never touched: comment lines, and the arguments of \erratum, \qslip, \ig,
\origpage, \fn, \includegraphics and the numeric arguments of \plateop/\pl/\tocl
— those quote the original exactly or are machine data.
"""
import re, glob, os, sys
SRC, OUT = "src/pages", "work/pages"
ACRONYMS = ["CSBC", "MSS", "SOAS", "BMS", "IOL", "IOR", "OUP", "SPG", "EIC",
"DCL", "FRS", "BEN", "MS", "BL"]
def build_maps():
plate, chap, subsec = {}, {}, {}
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
t = open(f).read()
for no, pg in re.findall(r"\\pl\{(\d+)\}\{.*?\}\{(\d+)\}", t):
plate[int(no)] = int(pg)
for no, pg in re.findall(r"\\tocl\{[^}]*\}\{(\d+)\}\{.*?\}\{(\d+)\}", t):
chap[int(no)] = int(pg)
m = re.search(r"\\subchap\{(\d+)\.([ivx]+)", t)
o = re.search(r"\\origpage\{(\d+)\}", t)
if m and o: subsec[f"{m.group(1)}{m.group(2)}"] = int(o.group(1))
m2 = re.search(r"\\chapnum\{Chapter (\d+)\}", t)
if m2 and o: chap.setdefault(int(m2.group(1)), int(o.group(1)))
return plate, chap, subsec
PLATE, CHAP, SUBSEC = build_maps()
def link(target, shown):
return r"\hyperlink{page.%d}{%s}" % (target, shown)
PROTECT = [
re.compile(r"\\(?:erratum|qslip)\{[^}]*\}(?:\{[^}]*\})?"),
re.compile(r"\\ig\{[^}]*\}"),
re.compile(r"\\includegraphics(?:\[[^\]]*\])?\{[^}]*\}"),
re.compile(r"\\(?:origpage|fn|hyperlink|hypertarget|pg)\{[^}]*\}"),
re.compile(r"\\(?:plateop|plate)\{[^}]*\}\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:pl|tocl|toclnp|erratumline)(?:\{[^}]*\})?"),
# heading macros: their arguments become PDF bookmarks and running-head
# marks, so they must stay plain text — a link inside them breaks the build
re.compile(r"\\(?:chapnum|partstart|sectionstart)\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:chapstart|chaphead|subchap)(?:\[[^\]]*\])?\{[^}]*\}"),
]
def protect(text):
store = []
def keep(m):
store.append(m.group(0)); return "\x00%d\x00" % (len(store)-1)
for rx in PROTECT: text = rx.sub(keep, text)
return text, store
def restore(text, store):
return re.sub(r"\x00(\d+)\x00", lambda m: store[int(m.group(1))], text)
def xrefs(t):
# pl. 57 / pls. 83 and 84 — always this thesis's own plates, always linked
def plate_sub(m):
head, nums = m.group(1), m.group(2)
def one(mm):
n = int(mm.group(0))
return link(PLATE[n], mm.group(0)) if n in PLATE else mm.group(0)
return head + "~" + re.sub(r"\d+", one, nums)
t = re.sub(r"\b(pls?\.)\s+((?:\d+)(?:\s*(?:,|and|-|--)\s*\d+)*)", plate_sub, t)
# p. 63 / pp. 196-7 — link only where the thesis cites ITSELF. Ross's own
# pages are introduced by a self-reference ("see above, p. 63", "mentioned
# above, see p. 60", a bare "see p. 43"); a page in somebody else's book sits
# in a citation — after a title, an "ibid.", a shelfmark, or publication data.
# External markers are tested first, because a citation may also contain
# "see". Anything matching neither is left alone: a missing link is cheaper
# than one that lands on the wrong page.
external = (r"\\emph\{[^}]*\}[^.]{0,40}$", r"\bibid\b", r"\bop\. ?cit",
r"\([^()]*\b\d{4}\b[^()]*\)[^.]{0,30}$", r"\b(?:Records|MSS?|IOR|IOL|BMS|SPG|CSBC|BL)\b[^;]{0,45}$",
r"\b(?:vol|nos?|pt)\.\s*[\dIVXL][^.]{0,20}$", r"\b[IVXL]{1,5},\s*$",
r"\bedn\b", r"\brpt\b", r"\bfacing\b",
# a surname followed by a comma, then title/edition matter: the
# commonest shape of a citation in this thesis ("Halhed,
# \emph{Grammar}, p. xxiii", "See Halhed, Grammar, pp. 57")
r"[A-Z][a-zA-Z']+,[^;]{0,45}$")
internal = (r"\babove\b", r"\bbelow\b", r"\bsee\b", r"\bchapters?\b",
r"\bdiscussed\b", r"\bmentioned\b", r"\bcited\b", r"\bstated\b")
def page_sub(m):
w = t[max(0, m.start()-130):m.start()]
if any(re.search(rx, w) for rx in external): return m.group(0)
if not any(re.search(rx, w, re.I) for rx in internal): return m.group(0)
head, nums = m.group(1), m.group(2)
first = re.match(r"\d+", nums)
if not first: return m.group(0)
n = int(first.group(0))
if n > 431: return m.group(0) # beyond this thesis's last page
return head + "~" + link(n, first.group(0)) + nums[first.end():].replace("-", "--")
t = re.sub(r"\b(pp?\.)\s+(\d+(?:\s*-\s*\d+)?)", page_sub, t)
# chapter 5 / chapters 7 and 8 / chapter 3ii — always self-references
def chap_sub(m):
head, rest = m.group(1), m.group(2)
def one(mm):
tok = mm.group(0)
if tok in SUBSEC: return link(SUBSEC[tok], tok)
d = re.match(r"\d+", tok)
return link(CHAP[int(d.group(0))], tok) if d and tok.isdigit() and int(d.group(0)) in CHAP else tok
return head + "~" + re.sub(r"\d+(?:[ivx]+)?", one, rest)
t = re.sub(r"\b(chapters?)\s+(\d+(?:[ivx]+)?(?:\s*(?:,|and)\s*\d+(?:[ivx]+)?)*)",
chap_sub, t, flags=re.I)
return t
TIE_WORDS = r"(?:no|nos|vol|vols|fig|figs|Mr|Mrs|Dr|St|Revd|Rev|pt|Pt)\."
def ties(t):
t = re.sub(r"\b(%s)\s+(?=[\dIVXL])" % TIE_WORDS, r"\1~", t)
t = re.sub(r"\b(Part|Section|Book|Vol|Plate|Table|Figure)\s+(?=[\dIVXL])", r"\1~", t)
t = re.sub(r"\b(\d+)\s+(lbs?|pt|pts|dpi|mm|cm|in)\b", r"\1~\2", t)
t = re.sub(r"\b([A-Z][a-z]+)\s+(I{1,3}V?|IV|VI{0,3}|IX|XI{0,2})\b", r"\1~\2", t)
return t
def endashes(t):
return re.sub(r"(?<=\d)-(?=\d)", "--", t)
def smallcaps(t):
for a in ACRONYMS:
t = re.sub(r"(?<![A-Za-z0-9\\])%s(?![A-Za-z0-9])" % a, r"\\textsc{%s}" % a.lower(), t)
return t
def mbox_specimens(t):
return re.sub(r"(\\ig\{[^}]*\})", r"\\mbox{\1}", t)
def polish(text):
out_lines = []
for line in text.split("\n"):
if line.lstrip().startswith("%"):
out_lines.append(line); continue
body, store = protect(line)
body = xrefs(body)
body = ties(body)
body = endashes(body)
body = smallcaps(body)
body = restore(body, store)
body = mbox_specimens(body)
out_lines.append(body)
return "\n".join(out_lines)
if __name__ == "__main__":
os.makedirs(OUT, exist_ok=True)
n = 0
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
open(os.path.join(OUT, os.path.basename(f)), "w").write(polish(open(f).read()))
n += 1
print(f"polished {n} pages -> {OUT}/ "
f"({len(PLATE)} plate targets, {len(CHAP)} chapters, {len(SUBSEC)} sections)")
Executable
+9
View File
@@ -0,0 +1,9 @@
#!/bin/sh
# prep.sh FIRST LAST — rasterize pages at 130 dpi for reading + tesseract scaffold text
cd "$(dirname "$0")/.."; mkdir -p scans work/ocr
pdftoppm -jpeg -r 130 -f "$1" -l "$2" "${ROSS_SRC:-source/10731406.pdf}" scans/hi
for p in $(seq "$1" "$2"); do
f=$(printf 'scans/hi-%03d.jpg' "$p")
tesseract "$f" "work/ocr/p$(printf '%04d' "$p")" -l eng --psm 4 >/dev/null 2>&1
done
ls scans/hi-*.jpg | tail -n +1 | wc -l
+153
View File
@@ -0,0 +1,153 @@
#!/usr/bin/env python3
r"""Reproduction check: every token in the transcription must reach the PDF.
The transcription in src/pages is the authority for what the edition should say;
the rendered PDF is what it does say. A token that appears N times in the source
and fewer than N times in the output is content the typesetter lost. Comparison
is by bag, not sequence, because endnotes and the Contents move text about.
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
"""
import re, sys, glob, subprocess, unicodedata
from collections import Counter
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
# macros whose text we keep, by which argument(s) carry copy
KEEP = {
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
"fn": [1], # \fn{n}{text} - n is a mark
"plateop": [3], "plate": [2], # caption only
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
"mbox": [0], "textsuperscript": [0],
}
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
def argspans(s, i):
"""yield (start, end) of consecutive brace groups beginning at i"""
out = []
while i < len(s) and s[i] in " \t":
i += 1
while i < len(s) and s[i] == "{":
d, j = 0, i
while j < len(s):
if s[j] == "\\":
j += 2; continue
if s[j] == "{": d += 1
elif s[j] == "}":
d -= 1
if d == 0:
out.append((i + 1, j)); i = j + 1; break
j += 1
else:
break
return out, i
def detex(s):
s = re.sub(r"(?<!\\)%.*", "", s)
out, i = [], 0
while i < len(s):
c = s[i]
if c != "\\":
out.append(c); i += 1; continue
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
if not m:
i += 2
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
k = re.match(r"\s*\[[^\]]*\]", s[i:])
if k:
i += k.end()
out.append(" "); continue # \& \% \\ ...
name = m.group(1)
j = i + m.end()
while True: # skip [optional] args
k = s.find("[", j)
if k != -1 and s[j:k].strip() == "":
e = s.find("]", k)
if e == -1:
break
j = e + 1
else:
break
spans, after = argspans(s, j)
if name in ("begin", "end"): # env name and column spec are not copy
spans = []
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
k = re.match(r"\s*[0-9]+", s[after:])
if k:
after += k.end()
if name in DROP:
out.append(" "); i = after; continue
keep = KEEP.get(name)
if keep is None:
out.append(" ")
for a, b in spans:
out.append(detex(s[a:b]) + " ")
else:
for k in keep:
if k < len(spans):
a, b = spans[k]
out.append(" " + detex(s[a:b]) + " ")
i = after
return "".join(out)
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
def tokens(text):
text = unicodedata.normalize("NFC", text)
for k, v in LIG.items():
text = text.replace(k, v)
text = text.replace("\u2019", "'").replace("\u2018", "'")
text = re.sub(r"[\u2010-\u2015]", "-", text)
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
src = []
files = sorted(glob.glob("src/pages/p*.tex"))
for f in files:
src.append(detex(open(f, encoding="utf-8").read()))
A = tokens("\n".join(src))
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
capture_output=True, text=True).stdout
B = tokens(pdftxt)
where = {}
for f, txt in zip(files, src):
for t in set(tokens(txt)):
where.setdefault(t, []).append(f[-8:-4])
# pdftotext merges two tokens wherever the gap is too small to register as a
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
# the extractor, not in the PDF, and they show up as a token that survives only
# inside a longer one - so report them apart instead of burying the real ones.
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
merged, real = [], []
for t in A:
if B[t] >= A[t]:
continue
row = (A[t] - B[t], t, A[t], B[t])
probe = unicodedata.normalize("NFD", t)
(merged if stream.count(probe) >= A[t] else real).append(row)
real.sort(key=lambda r: (-r[0], r[1]))
merged.sort(key=lambda r: (-r[0], r[1]))
deficits = real
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
f"({len(merged)} more merged into a neighbour by the extractor)")
for n, t, a, b in deficits:
pp = where.get(t, [])
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
if merged:
print("extractor merges (not losses):",
", ".join(f"{t}(-{n})" for n, t, a, b in merged))
+17
View File
@@ -0,0 +1,17 @@
#!/usr/bin/env python3
"""setrow.py PDFPAGE PRINTED KIND NOTES — mark one manifest.tsv row done.
python3 tools/setrow.py 20 18 plate "plate 1"
src_file is derived from the PDF page (blank for kind dropped/colophon)."""
import sys
pdf, printed, kind, notes = sys.argv[1:5]
rows = open("manifest.tsv").read().splitlines()
out = []
for i, line in enumerate(rows):
c = line.split("\t")
if i and c[0] == pdf:
src = "" if kind in ("dropped", "colophon") else f"src/pages/p{int(pdf):04d}.tex"
line = "\t".join([c[0], c[1], printed, kind, "done", src, notes])
out.append(line)
open("manifest.tsv", "w").write("\n".join(out) + "\n")
print(" | ".join(out[int(pdf)].split("\t")))