CLAUDE.md requires re-auditing the cross-reference rule after any change to it, and polish.py changed twice this session. tools/linkaudit.py does it: three classes checked mechanically, the fourth printed to be read. dangling targets 0 plate page mismatch (listed vs set) 0 (all 178) chapter page mismatch 0 `p. N` self-references 25 all read, all sound The finding: an elided range end is not a plate number of its own. plate_sub ran re.sub(r"\d+") over the whole range, so "pls. 146-8" linked its 8 to plate 8 (orig. p.35) and "pls. 142-5" linked its 5 to plate 5 (p.23). Three links, all pointing at the wrong plate — invisible to every other check, since the targets exist and the text is correct. The end is now expanded against the leading digits of the number it is elided against: 146-8 -> plate 148 (p.334), 142-5 -> plate 145 (p.329), both confirmed against the List of Plates and against where \plateop sets them. Worth noting what the audit also confirmed: in two footnotes the rule discriminates a citation page from a self-reference in the same sentence — "(London, 1962), p. 43; also see below, p. 63" links only the second. NOT YET BUILT: the Docker daemon is down, so make verify has not run against this change. polish.py was run on the host and the audit re-run against its output, but the PDF does not yet carry the fix. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
196 lines
9.3 KiB
Python
Executable File
196 lines
9.3 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
r"""polish.py — build-time typographic pass over the transcription.
|
|
|
|
Reads src/pages/*.tex and writes work/pages/*.tex. The transcription itself stays
|
|
exactly as typed; this pass applies at set time the typography the author's
|
|
typewriter could not produce, plus the cross-reference links a digital edition
|
|
should have:
|
|
|
|
* ties, so a reference or a unit never breaks across a line
|
|
* en-dashes for numeric ranges (1778-1978, pp. 315-320)
|
|
* (no small caps: see smallcaps() below — the pass was removed 2026-09-15)
|
|
* p./pp., pl./pls. and chapter references become links to the original page
|
|
they name — note that the original's page numbers are the targets, not the
|
|
edition's, since `\origpage{N}` plants `page.N` where original page N begins
|
|
* \mbox round an inline specimen so it never breaks from adjacent punctuation
|
|
|
|
Never touched: comment lines, and the arguments of \erratum, \qslip, \ig,
|
|
\origpage, \fn, \includegraphics and the numeric arguments of \plateop/\pl/\tocl
|
|
— those quote the original exactly or are machine data.
|
|
"""
|
|
import re, glob, os, sys
|
|
|
|
SRC, OUT = "src/pages", "work/pages"
|
|
|
|
# Kept only as a record of what the removed pass covered. See smallcaps().
|
|
_FORMER_ACRONYMS = ["CSBC", "MSS", "SOAS", "BMS", "IOL", "IOR", "OUP", "SPG",
|
|
"EIC", "DCL", "FRS", "BEN", "MS", "BL"]
|
|
|
|
def build_maps():
|
|
plate, chap, subsec = {}, {}, {}
|
|
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
|
|
t = open(f).read()
|
|
for no, pg in re.findall(r"\\pl\{(\d+)\}\{.*?\}\{(\d+)\}", t):
|
|
plate[int(no)] = int(pg)
|
|
# \plx is the one List-of-Plates entry whose location is not a bare page
|
|
# number (plate 1 is "page 18", a plate facing the text), so its page
|
|
# reference is wrapped in \pg. Without this, plate 1 was the single plate
|
|
# whose in-text "pl. 1" references never linked.
|
|
for no, pg in re.findall(r"\\plx\{(\d+)\}\{.*?\}\{.*?\\pg\{(\d+)\}", t):
|
|
plate[int(no)] = int(pg)
|
|
for no, pg in re.findall(r"\\tocl\{[^}]*\}\{(\d+)\}\{.*?\}\{(\d+)\}", t):
|
|
chap[int(no)] = int(pg)
|
|
m = re.search(r"\\subchap\{(\d+)\.([ivx]+)", t)
|
|
o = re.search(r"\\origpage\{(\d+)\}", t)
|
|
if m and o: subsec[f"{m.group(1)}{m.group(2)}"] = int(o.group(1))
|
|
m2 = re.search(r"\\chapnum\{Chapter (\d+)\}", t)
|
|
if m2 and o: chap.setdefault(int(m2.group(1)), int(o.group(1)))
|
|
return plate, chap, subsec
|
|
|
|
PLATE, CHAP, SUBSEC = build_maps()
|
|
|
|
def link(target, shown):
|
|
# \xref, not \hyperlink: the preamble gives \xref a visible hairline rule,
|
|
# while \hyperlink stays unmarked for the whole-line Contents and plate entries.
|
|
return r"\xref{page.%d}{%s}" % (target, shown)
|
|
|
|
PROTECT = [
|
|
re.compile(r"\\(?:erratum|qslip)\{[^}]*\}(?:\{[^}]*\})?"),
|
|
re.compile(r"\\ig\{[^}]*\}"),
|
|
re.compile(r"\\includegraphics(?:\[[^\]]*\])?\{[^}]*\}"),
|
|
re.compile(r"\\(?:origpage|fn|hyperlink|hypertarget|pg|xref)\{[^}]*\}"),
|
|
re.compile(r"\\(?:plateop|plate)\{[^}]*\}\{[^}]*\}\{[^}]*\}"),
|
|
re.compile(r"\\(?:pl|tocl|toclnp|erratumline)(?:\{[^}]*\})?"),
|
|
# heading macros: their arguments become PDF bookmarks and running-head
|
|
# marks, so they must stay plain text — a link inside them breaks the build
|
|
re.compile(r"\\(?:chapnum|partstart|sectionstart)\{[^}]*\}\{[^}]*\}"),
|
|
re.compile(r"\\(?:chapstart|chaphead|subchap)(?:\[[^\]]*\])?\{[^}]*\}"),
|
|
]
|
|
|
|
def protect(text):
|
|
store = []
|
|
def keep(m):
|
|
store.append(m.group(0)); return "\x00%d\x00" % (len(store)-1)
|
|
for rx in PROTECT: text = rx.sub(keep, text)
|
|
return text, store
|
|
|
|
def restore(text, store):
|
|
return re.sub(r"\x00(\d+)\x00", lambda m: store[int(m.group(1))], text)
|
|
|
|
def xrefs(t):
|
|
# pl. 57 / pls. 83 and 84 — always this thesis's own plates, always linked
|
|
def plate_sub(m):
|
|
head, nums = m.group(1), m.group(2)
|
|
# An elided range end ("pls. 146-8" for 146-148, "142-5" for 142-145) is
|
|
# NOT a plate number of its own. Linking it as one sent "pls. 146-8" to
|
|
# plate 8 and "pls. 142-5" to plate 5 — three wrong links, found by the
|
|
# link audit 2026-09-15. Expand it from the leading digits of the number
|
|
# it is elided against before looking the plate up.
|
|
prev = None
|
|
def one(mm):
|
|
nonlocal prev
|
|
tok = mm.group(0); n = int(tok)
|
|
if prev is not None and n < prev:
|
|
full = int(str(prev)[:len(str(prev)) - len(tok)] + tok)
|
|
return link(PLATE[full], tok) if full in PLATE else tok
|
|
prev = n
|
|
return link(PLATE[n], tok) if n in PLATE else tok
|
|
return head + "~" + re.sub(r"\d+", one, nums)
|
|
t = re.sub(r"\b(pls?\.)\s+((?:\d+)(?:\s*(?:,|and|-|--)\s*\d+)*)", plate_sub, t)
|
|
|
|
# p. 63 / pp. 196-7 — link only where the thesis cites ITSELF. Ross's own
|
|
# pages are introduced by a self-reference ("see above, p. 63", "mentioned
|
|
# above, see p. 60", a bare "see p. 43"); a page in somebody else's book sits
|
|
# in a citation — after a title, an "ibid.", a shelfmark, or publication data.
|
|
# External markers are tested first, because a citation may also contain
|
|
# "see". Anything matching neither is left alone: a missing link is cheaper
|
|
# than one that lands on the wrong page.
|
|
external = (r"\\emph\{[^}]*\}[^.]{0,40}$", r"\bibid\b", r"\bop\. ?cit",
|
|
r"\([^()]*\b\d{4}\b[^()]*\)[^.]{0,30}$", r"\b(?:Records|MSS?|IOR|IOL|BMS|SPG|CSBC|BL)\b[^;]{0,45}$",
|
|
r"\b(?:vol|nos?|pt)\.\s*[\dIVXL][^.]{0,20}$", r"\b[IVXL]{1,5},\s*$",
|
|
r"\bedn\b", r"\brpt\b", r"\bfacing\b",
|
|
# a surname followed by a comma, then title/edition matter: the
|
|
# commonest shape of a citation in this thesis ("Halhed,
|
|
# \emph{Grammar}, p. xxiii", "See Halhed, Grammar, pp. 57")
|
|
r"[A-Z][a-zA-Z']+,[^;]{0,45}$")
|
|
internal = (r"\babove\b", r"\bbelow\b", r"\bsee\b", r"\bchapters?\b",
|
|
r"\bdiscussed\b", r"\bmentioned\b", r"\bcited\b", r"\bstated\b")
|
|
def page_sub(m):
|
|
w = t[max(0, m.start()-130):m.start()]
|
|
if any(re.search(rx, w) for rx in external): return m.group(0)
|
|
if not any(re.search(rx, w, re.I) for rx in internal): return m.group(0)
|
|
head, nums = m.group(1), m.group(2)
|
|
first = re.match(r"\d+", nums)
|
|
if not first: return m.group(0)
|
|
n = int(first.group(0))
|
|
if n > 431: return m.group(0) # beyond this thesis's last page
|
|
return head + "~" + link(n, first.group(0)) + nums[first.end():].replace("-", "--")
|
|
t = re.sub(r"\b(pp?\.)\s+(\d+(?:\s*-\s*\d+)?)", page_sub, t)
|
|
|
|
# chapter 5 / chapters 7 and 8 / chapter 3ii — always self-references
|
|
def chap_sub(m):
|
|
head, rest = m.group(1), m.group(2)
|
|
def one(mm):
|
|
tok = mm.group(0)
|
|
if tok in SUBSEC: return link(SUBSEC[tok], tok)
|
|
d = re.match(r"\d+", tok)
|
|
return link(CHAP[int(d.group(0))], tok) if d and tok.isdigit() and int(d.group(0)) in CHAP else tok
|
|
return head + "~" + re.sub(r"\d+(?:[ivx]+)?", one, rest)
|
|
t = re.sub(r"\b(chapters?)\s+(\d+(?:[ivx]+)?(?:\s*(?:,|and)\s*\d+(?:[ivx]+)?)*)",
|
|
chap_sub, t, flags=re.I)
|
|
return t
|
|
|
|
TIE_WORDS = r"(?:no|nos|vol|vols|fig|figs|Mr|Mrs|Dr|St|Revd|Rev|pt|Pt)\."
|
|
def ties(t):
|
|
t = re.sub(r"\b(%s)\s+(?=[\dIVXL])" % TIE_WORDS, r"\1~", t)
|
|
t = re.sub(r"\b(Part|Section|Book|Vol|Plate|Table|Figure)\s+(?=[\dIVXL])", r"\1~", t)
|
|
t = re.sub(r"\b(\d+)\s+(lbs?|pt|pts|dpi|mm|cm|in)\b", r"\1~\2", t)
|
|
t = re.sub(r"\b([A-Z][a-z]+)\s+(I{1,3}V?|IV|VI{0,3}|IX|XI{0,2})\b", r"\1~\2", t)
|
|
return t
|
|
|
|
def endashes(t):
|
|
return re.sub(r"(?<=\d)-(?=\d)", "--", t)
|
|
|
|
def smallcaps(t):
|
|
"""No-op since 2026-09-15 (Sammay): abbreviations stay plain uppercase.
|
|
|
|
The pass small-capped an explicit whitelist of institutional acronyms. That
|
|
whitelist could only ever be partial, and on the Abbreviations and
|
|
Conventions page the gap was plain: BFBS, LMS and MLCo sat in one column
|
|
against small-capped BL, BMS, EIC, IOL, OUP, SOAS and SPG. It also split a
|
|
shelfmark, setting a small-cap MS against a full-cap EUR in "MS EUR 30".
|
|
|
|
Completing the list was the obvious fix and is the wrong one. The original
|
|
is a typescript: a typewriter cannot set small caps, so the source defines
|
|
no abbreviation in them and every acronym on the page is plain uppercase.
|
|
Small caps were this edition's invention, and an inconsistent one. Rule 3 —
|
|
the original's typography stands — settles it."""
|
|
return t
|
|
|
|
def mbox_specimens(t):
|
|
return re.sub(r"(\\ig\{[^}]*\})", r"\\mbox{\1}", t)
|
|
|
|
def polish(text):
|
|
out_lines = []
|
|
for line in text.split("\n"):
|
|
if line.lstrip().startswith("%"):
|
|
out_lines.append(line); continue
|
|
body, store = protect(line)
|
|
body = xrefs(body)
|
|
body = ties(body)
|
|
body = endashes(body)
|
|
body = smallcaps(body)
|
|
body = restore(body, store)
|
|
body = mbox_specimens(body)
|
|
out_lines.append(body)
|
|
return "\n".join(out_lines)
|
|
|
|
if __name__ == "__main__":
|
|
os.makedirs(OUT, exist_ok=True)
|
|
n = 0
|
|
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
|
|
open(os.path.join(OUT, os.path.basename(f)), "w").write(polish(open(f).read()))
|
|
n += 1
|
|
print(f"polished {n} pages -> {OUT}/ "
|
|
f"({len(PLATE)} plate targets, {len(CHAP)} chapters, {len(SUBSEC)} sections)")
|