Files
tepbc-ross-retyped/tools/polish.py
T
bdeshiandClaude Opus 5 82849eb94d Initial commit: Ross 1988 re-typeset edition
A searchable, re-typeset edition of Fiona Ross, "The Evolution of the
Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988),
transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178
plates and 410 inline type specimens cut from the scan; 51 errata.

Tracked: the transcription (src/pages), the preamble and its typographic
decisions, the cut images (plates/ — not reliably regenerable, the crop
specs for the inline cuts were never scripted), tools, and the four
working documents.

Not tracked: the built PDF, which `make` remakes from src/ and plates/;
the ProQuest scan under source/, which is third-party and needed only by
`make prep` and `make plate`; scans/ and work/, both regenerable.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 19:03:25 +06:00

165 lines
7.3 KiB
Python
Executable File

#!/usr/bin/env python3
r"""polish.py — build-time typographic pass over the transcription.
Reads src/pages/*.tex and writes work/pages/*.tex. The transcription itself stays
exactly as typed; this pass applies at set time the typography the author's
typewriter could not produce, plus the cross-reference links a digital edition
should have:
* ties, so a reference or a unit never breaks across a line
* en-dashes for numeric ranges (1778-1978, pp. 315-320)
* small caps for institutional acronyms (explicit whitelist below)
* p./pp., pl./pls. and chapter references become links to the original page
they name — note that the original's page numbers are the targets, not the
edition's, since `\origpage{N}` plants `page.N` where original page N begins
* \mbox round an inline specimen so it never breaks from adjacent punctuation
Never touched: comment lines, and the arguments of \erratum, \qslip, \ig,
\origpage, \fn, \includegraphics and the numeric arguments of \plateop/\pl/\tocl
— those quote the original exactly or are machine data.
"""
import re, glob, os, sys
SRC, OUT = "src/pages", "work/pages"
ACRONYMS = ["CSBC", "MSS", "SOAS", "BMS", "IOL", "IOR", "OUP", "SPG", "EIC",
"DCL", "FRS", "BEN", "MS", "BL"]
def build_maps():
plate, chap, subsec = {}, {}, {}
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
t = open(f).read()
for no, pg in re.findall(r"\\pl\{(\d+)\}\{.*?\}\{(\d+)\}", t):
plate[int(no)] = int(pg)
for no, pg in re.findall(r"\\tocl\{[^}]*\}\{(\d+)\}\{.*?\}\{(\d+)\}", t):
chap[int(no)] = int(pg)
m = re.search(r"\\subchap\{(\d+)\.([ivx]+)", t)
o = re.search(r"\\origpage\{(\d+)\}", t)
if m and o: subsec[f"{m.group(1)}{m.group(2)}"] = int(o.group(1))
m2 = re.search(r"\\chapnum\{Chapter (\d+)\}", t)
if m2 and o: chap.setdefault(int(m2.group(1)), int(o.group(1)))
return plate, chap, subsec
PLATE, CHAP, SUBSEC = build_maps()
def link(target, shown):
return r"\hyperlink{page.%d}{%s}" % (target, shown)
PROTECT = [
re.compile(r"\\(?:erratum|qslip)\{[^}]*\}(?:\{[^}]*\})?"),
re.compile(r"\\ig\{[^}]*\}"),
re.compile(r"\\includegraphics(?:\[[^\]]*\])?\{[^}]*\}"),
re.compile(r"\\(?:origpage|fn|hyperlink|hypertarget|pg)\{[^}]*\}"),
re.compile(r"\\(?:plateop|plate)\{[^}]*\}\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:pl|tocl|toclnp|erratumline)(?:\{[^}]*\})?"),
# heading macros: their arguments become PDF bookmarks and running-head
# marks, so they must stay plain text — a link inside them breaks the build
re.compile(r"\\(?:chapnum|partstart|sectionstart)\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:chapstart|chaphead|subchap)(?:\[[^\]]*\])?\{[^}]*\}"),
]
def protect(text):
store = []
def keep(m):
store.append(m.group(0)); return "\x00%d\x00" % (len(store)-1)
for rx in PROTECT: text = rx.sub(keep, text)
return text, store
def restore(text, store):
return re.sub(r"\x00(\d+)\x00", lambda m: store[int(m.group(1))], text)
def xrefs(t):
# pl. 57 / pls. 83 and 84 — always this thesis's own plates, always linked
def plate_sub(m):
head, nums = m.group(1), m.group(2)
def one(mm):
n = int(mm.group(0))
return link(PLATE[n], mm.group(0)) if n in PLATE else mm.group(0)
return head + "~" + re.sub(r"\d+", one, nums)
t = re.sub(r"\b(pls?\.)\s+((?:\d+)(?:\s*(?:,|and|-|--)\s*\d+)*)", plate_sub, t)
# p. 63 / pp. 196-7 — link only where the thesis cites ITSELF. Ross's own
# pages are introduced by a self-reference ("see above, p. 63", "mentioned
# above, see p. 60", a bare "see p. 43"); a page in somebody else's book sits
# in a citation — after a title, an "ibid.", a shelfmark, or publication data.
# External markers are tested first, because a citation may also contain
# "see". Anything matching neither is left alone: a missing link is cheaper
# than one that lands on the wrong page.
external = (r"\\emph\{[^}]*\}[^.]{0,40}$", r"\bibid\b", r"\bop\. ?cit",
r"\([^()]*\b\d{4}\b[^()]*\)[^.]{0,30}$", r"\b(?:Records|MSS?|IOR|IOL|BMS|SPG|CSBC|BL)\b[^;]{0,45}$",
r"\b(?:vol|nos?|pt)\.\s*[\dIVXL][^.]{0,20}$", r"\b[IVXL]{1,5},\s*$",
r"\bedn\b", r"\brpt\b", r"\bfacing\b",
# a surname followed by a comma, then title/edition matter: the
# commonest shape of a citation in this thesis ("Halhed,
# \emph{Grammar}, p. xxiii", "See Halhed, Grammar, pp. 57")
r"[A-Z][a-zA-Z']+,[^;]{0,45}$")
internal = (r"\babove\b", r"\bbelow\b", r"\bsee\b", r"\bchapters?\b",
r"\bdiscussed\b", r"\bmentioned\b", r"\bcited\b", r"\bstated\b")
def page_sub(m):
w = t[max(0, m.start()-130):m.start()]
if any(re.search(rx, w) for rx in external): return m.group(0)
if not any(re.search(rx, w, re.I) for rx in internal): return m.group(0)
head, nums = m.group(1), m.group(2)
first = re.match(r"\d+", nums)
if not first: return m.group(0)
n = int(first.group(0))
if n > 431: return m.group(0) # beyond this thesis's last page
return head + "~" + link(n, first.group(0)) + nums[first.end():].replace("-", "--")
t = re.sub(r"\b(pp?\.)\s+(\d+(?:\s*-\s*\d+)?)", page_sub, t)
# chapter 5 / chapters 7 and 8 / chapter 3ii — always self-references
def chap_sub(m):
head, rest = m.group(1), m.group(2)
def one(mm):
tok = mm.group(0)
if tok in SUBSEC: return link(SUBSEC[tok], tok)
d = re.match(r"\d+", tok)
return link(CHAP[int(d.group(0))], tok) if d and tok.isdigit() and int(d.group(0)) in CHAP else tok
return head + "~" + re.sub(r"\d+(?:[ivx]+)?", one, rest)
t = re.sub(r"\b(chapters?)\s+(\d+(?:[ivx]+)?(?:\s*(?:,|and)\s*\d+(?:[ivx]+)?)*)",
chap_sub, t, flags=re.I)
return t
TIE_WORDS = r"(?:no|nos|vol|vols|fig|figs|Mr|Mrs|Dr|St|Revd|Rev|pt|Pt)\."
def ties(t):
t = re.sub(r"\b(%s)\s+(?=[\dIVXL])" % TIE_WORDS, r"\1~", t)
t = re.sub(r"\b(Part|Section|Book|Vol|Plate|Table|Figure)\s+(?=[\dIVXL])", r"\1~", t)
t = re.sub(r"\b(\d+)\s+(lbs?|pt|pts|dpi|mm|cm|in)\b", r"\1~\2", t)
t = re.sub(r"\b([A-Z][a-z]+)\s+(I{1,3}V?|IV|VI{0,3}|IX|XI{0,2})\b", r"\1~\2", t)
return t
def endashes(t):
return re.sub(r"(?<=\d)-(?=\d)", "--", t)
def smallcaps(t):
for a in ACRONYMS:
t = re.sub(r"(?<![A-Za-z0-9\\])%s(?![A-Za-z0-9])" % a, r"\\textsc{%s}" % a.lower(), t)
return t
def mbox_specimens(t):
return re.sub(r"(\\ig\{[^}]*\})", r"\\mbox{\1}", t)
def polish(text):
out_lines = []
for line in text.split("\n"):
if line.lstrip().startswith("%"):
out_lines.append(line); continue
body, store = protect(line)
body = xrefs(body)
body = ties(body)
body = endashes(body)
body = smallcaps(body)
body = restore(body, store)
body = mbox_specimens(body)
out_lines.append(body)
return "\n".join(out_lines)
if __name__ == "__main__":
os.makedirs(OUT, exist_ok=True)
n = 0
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
open(os.path.join(OUT, os.path.basename(f)), "w").write(polish(open(f).read()))
n += 1
print(f"polished {n} pages -> {OUT}/ "
f"({len(PLATE)} plate targets, {len(CHAP)} chapters, {len(SUBSEC)} sections)")