A searchable, re-typeset edition of Fiona Ross, "The Evolution of the Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988), transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178 plates and 410 inline type specimens cut from the scan; 51 errata. Tracked: the transcription (src/pages), the preamble and its typographic decisions, the cut images (plates/ — not reliably regenerable, the crop specs for the inline cuts were never scripted), tools, and the four working documents. Not tracked: the built PDF, which `make` remakes from src/ and plates/; the ProQuest scan under source/, which is third-party and needed only by `make prep` and `make plate`; scans/ and work/, both regenerable. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
165 lines
7.3 KiB
Python
Executable File
165 lines
7.3 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
r"""polish.py — build-time typographic pass over the transcription.
|
|
|
|
Reads src/pages/*.tex and writes work/pages/*.tex. The transcription itself stays
|
|
exactly as typed; this pass applies at set time the typography the author's
|
|
typewriter could not produce, plus the cross-reference links a digital edition
|
|
should have:
|
|
|
|
* ties, so a reference or a unit never breaks across a line
|
|
* en-dashes for numeric ranges (1778-1978, pp. 315-320)
|
|
* small caps for institutional acronyms (explicit whitelist below)
|
|
* p./pp., pl./pls. and chapter references become links to the original page
|
|
they name — note that the original's page numbers are the targets, not the
|
|
edition's, since `\origpage{N}` plants `page.N` where original page N begins
|
|
* \mbox round an inline specimen so it never breaks from adjacent punctuation
|
|
|
|
Never touched: comment lines, and the arguments of \erratum, \qslip, \ig,
|
|
\origpage, \fn, \includegraphics and the numeric arguments of \plateop/\pl/\tocl
|
|
— those quote the original exactly or are machine data.
|
|
"""
|
|
import re, glob, os, sys
|
|
|
|
SRC, OUT = "src/pages", "work/pages"
|
|
|
|
ACRONYMS = ["CSBC", "MSS", "SOAS", "BMS", "IOL", "IOR", "OUP", "SPG", "EIC",
|
|
"DCL", "FRS", "BEN", "MS", "BL"]
|
|
|
|
def build_maps():
|
|
plate, chap, subsec = {}, {}, {}
|
|
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
|
|
t = open(f).read()
|
|
for no, pg in re.findall(r"\\pl\{(\d+)\}\{.*?\}\{(\d+)\}", t):
|
|
plate[int(no)] = int(pg)
|
|
for no, pg in re.findall(r"\\tocl\{[^}]*\}\{(\d+)\}\{.*?\}\{(\d+)\}", t):
|
|
chap[int(no)] = int(pg)
|
|
m = re.search(r"\\subchap\{(\d+)\.([ivx]+)", t)
|
|
o = re.search(r"\\origpage\{(\d+)\}", t)
|
|
if m and o: subsec[f"{m.group(1)}{m.group(2)}"] = int(o.group(1))
|
|
m2 = re.search(r"\\chapnum\{Chapter (\d+)\}", t)
|
|
if m2 and o: chap.setdefault(int(m2.group(1)), int(o.group(1)))
|
|
return plate, chap, subsec
|
|
|
|
PLATE, CHAP, SUBSEC = build_maps()
|
|
|
|
def link(target, shown):
|
|
return r"\hyperlink{page.%d}{%s}" % (target, shown)
|
|
|
|
PROTECT = [
|
|
re.compile(r"\\(?:erratum|qslip)\{[^}]*\}(?:\{[^}]*\})?"),
|
|
re.compile(r"\\ig\{[^}]*\}"),
|
|
re.compile(r"\\includegraphics(?:\[[^\]]*\])?\{[^}]*\}"),
|
|
re.compile(r"\\(?:origpage|fn|hyperlink|hypertarget|pg)\{[^}]*\}"),
|
|
re.compile(r"\\(?:plateop|plate)\{[^}]*\}\{[^}]*\}\{[^}]*\}"),
|
|
re.compile(r"\\(?:pl|tocl|toclnp|erratumline)(?:\{[^}]*\})?"),
|
|
# heading macros: their arguments become PDF bookmarks and running-head
|
|
# marks, so they must stay plain text — a link inside them breaks the build
|
|
re.compile(r"\\(?:chapnum|partstart|sectionstart)\{[^}]*\}\{[^}]*\}"),
|
|
re.compile(r"\\(?:chapstart|chaphead|subchap)(?:\[[^\]]*\])?\{[^}]*\}"),
|
|
]
|
|
|
|
def protect(text):
|
|
store = []
|
|
def keep(m):
|
|
store.append(m.group(0)); return "\x00%d\x00" % (len(store)-1)
|
|
for rx in PROTECT: text = rx.sub(keep, text)
|
|
return text, store
|
|
|
|
def restore(text, store):
|
|
return re.sub(r"\x00(\d+)\x00", lambda m: store[int(m.group(1))], text)
|
|
|
|
def xrefs(t):
|
|
# pl. 57 / pls. 83 and 84 — always this thesis's own plates, always linked
|
|
def plate_sub(m):
|
|
head, nums = m.group(1), m.group(2)
|
|
def one(mm):
|
|
n = int(mm.group(0))
|
|
return link(PLATE[n], mm.group(0)) if n in PLATE else mm.group(0)
|
|
return head + "~" + re.sub(r"\d+", one, nums)
|
|
t = re.sub(r"\b(pls?\.)\s+((?:\d+)(?:\s*(?:,|and|-|--)\s*\d+)*)", plate_sub, t)
|
|
|
|
# p. 63 / pp. 196-7 — link only where the thesis cites ITSELF. Ross's own
|
|
# pages are introduced by a self-reference ("see above, p. 63", "mentioned
|
|
# above, see p. 60", a bare "see p. 43"); a page in somebody else's book sits
|
|
# in a citation — after a title, an "ibid.", a shelfmark, or publication data.
|
|
# External markers are tested first, because a citation may also contain
|
|
# "see". Anything matching neither is left alone: a missing link is cheaper
|
|
# than one that lands on the wrong page.
|
|
external = (r"\\emph\{[^}]*\}[^.]{0,40}$", r"\bibid\b", r"\bop\. ?cit",
|
|
r"\([^()]*\b\d{4}\b[^()]*\)[^.]{0,30}$", r"\b(?:Records|MSS?|IOR|IOL|BMS|SPG|CSBC|BL)\b[^;]{0,45}$",
|
|
r"\b(?:vol|nos?|pt)\.\s*[\dIVXL][^.]{0,20}$", r"\b[IVXL]{1,5},\s*$",
|
|
r"\bedn\b", r"\brpt\b", r"\bfacing\b",
|
|
# a surname followed by a comma, then title/edition matter: the
|
|
# commonest shape of a citation in this thesis ("Halhed,
|
|
# \emph{Grammar}, p. xxiii", "See Halhed, Grammar, pp. 57")
|
|
r"[A-Z][a-zA-Z']+,[^;]{0,45}$")
|
|
internal = (r"\babove\b", r"\bbelow\b", r"\bsee\b", r"\bchapters?\b",
|
|
r"\bdiscussed\b", r"\bmentioned\b", r"\bcited\b", r"\bstated\b")
|
|
def page_sub(m):
|
|
w = t[max(0, m.start()-130):m.start()]
|
|
if any(re.search(rx, w) for rx in external): return m.group(0)
|
|
if not any(re.search(rx, w, re.I) for rx in internal): return m.group(0)
|
|
head, nums = m.group(1), m.group(2)
|
|
first = re.match(r"\d+", nums)
|
|
if not first: return m.group(0)
|
|
n = int(first.group(0))
|
|
if n > 431: return m.group(0) # beyond this thesis's last page
|
|
return head + "~" + link(n, first.group(0)) + nums[first.end():].replace("-", "--")
|
|
t = re.sub(r"\b(pp?\.)\s+(\d+(?:\s*-\s*\d+)?)", page_sub, t)
|
|
|
|
# chapter 5 / chapters 7 and 8 / chapter 3ii — always self-references
|
|
def chap_sub(m):
|
|
head, rest = m.group(1), m.group(2)
|
|
def one(mm):
|
|
tok = mm.group(0)
|
|
if tok in SUBSEC: return link(SUBSEC[tok], tok)
|
|
d = re.match(r"\d+", tok)
|
|
return link(CHAP[int(d.group(0))], tok) if d and tok.isdigit() and int(d.group(0)) in CHAP else tok
|
|
return head + "~" + re.sub(r"\d+(?:[ivx]+)?", one, rest)
|
|
t = re.sub(r"\b(chapters?)\s+(\d+(?:[ivx]+)?(?:\s*(?:,|and)\s*\d+(?:[ivx]+)?)*)",
|
|
chap_sub, t, flags=re.I)
|
|
return t
|
|
|
|
TIE_WORDS = r"(?:no|nos|vol|vols|fig|figs|Mr|Mrs|Dr|St|Revd|Rev|pt|Pt)\."
|
|
def ties(t):
|
|
t = re.sub(r"\b(%s)\s+(?=[\dIVXL])" % TIE_WORDS, r"\1~", t)
|
|
t = re.sub(r"\b(Part|Section|Book|Vol|Plate|Table|Figure)\s+(?=[\dIVXL])", r"\1~", t)
|
|
t = re.sub(r"\b(\d+)\s+(lbs?|pt|pts|dpi|mm|cm|in)\b", r"\1~\2", t)
|
|
t = re.sub(r"\b([A-Z][a-z]+)\s+(I{1,3}V?|IV|VI{0,3}|IX|XI{0,2})\b", r"\1~\2", t)
|
|
return t
|
|
|
|
def endashes(t):
|
|
return re.sub(r"(?<=\d)-(?=\d)", "--", t)
|
|
|
|
def smallcaps(t):
|
|
for a in ACRONYMS:
|
|
t = re.sub(r"(?<![A-Za-z0-9\\])%s(?![A-Za-z0-9])" % a, r"\\textsc{%s}" % a.lower(), t)
|
|
return t
|
|
|
|
def mbox_specimens(t):
|
|
return re.sub(r"(\\ig\{[^}]*\})", r"\\mbox{\1}", t)
|
|
|
|
def polish(text):
|
|
out_lines = []
|
|
for line in text.split("\n"):
|
|
if line.lstrip().startswith("%"):
|
|
out_lines.append(line); continue
|
|
body, store = protect(line)
|
|
body = xrefs(body)
|
|
body = ties(body)
|
|
body = endashes(body)
|
|
body = smallcaps(body)
|
|
body = restore(body, store)
|
|
body = mbox_specimens(body)
|
|
out_lines.append(body)
|
|
return "\n".join(out_lines)
|
|
|
|
if __name__ == "__main__":
|
|
os.makedirs(OUT, exist_ok=True)
|
|
n = 0
|
|
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
|
|
open(os.path.join(OUT, os.path.basename(f)), "w").write(polish(open(f).read()))
|
|
n += 1
|
|
print(f"polished {n} pages -> {OUT}/ "
|
|
f"({len(PLATE)} plate targets, {len(CHAP)} chapters, {len(SUBSEC)} sections)")
|