Files
bdeshiandClaude Opus 5 e602e7e631 Link audit: fix three plate links from elided ranges
CLAUDE.md requires re-auditing the cross-reference rule after any change
to it, and polish.py changed twice this session. tools/linkaudit.py does
it: three classes checked mechanically, the fourth printed to be read.

  dangling targets                      0
  plate page mismatch (listed vs set)   0   (all 178)
  chapter page mismatch                 0
  `p. N` self-references               25   all read, all sound

The finding: an elided range end is not a plate number of its own.
plate_sub ran re.sub(r"\d+") over the whole range, so "pls. 146-8" linked
its 8 to plate 8 (orig. p.35) and "pls. 142-5" linked its 5 to plate 5
(p.23). Three links, all pointing at the wrong plate — invisible to every
other check, since the targets exist and the text is correct. The end is
now expanded against the leading digits of the number it is elided
against: 146-8 -> plate 148 (p.334), 142-5 -> plate 145 (p.329), both
confirmed against the List of Plates and against where \plateop sets
them.

Worth noting what the audit also confirmed: in two footnotes the rule
discriminates a citation page from a self-reference in the same sentence
— "(London, 1962), p. 43; also see below, p. 63" links only the second.

NOT YET BUILT: the Docker daemon is down, so make verify has not run
against this change. polish.py was run on the host and the audit re-run
against its output, but the PDF does not yet carry the fix.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 10:00:13 +06:00

196 lines
9.3 KiB
Python
Executable File

#!/usr/bin/env python3
r"""polish.py — build-time typographic pass over the transcription.
Reads src/pages/*.tex and writes work/pages/*.tex. The transcription itself stays
exactly as typed; this pass applies at set time the typography the author's
typewriter could not produce, plus the cross-reference links a digital edition
should have:
* ties, so a reference or a unit never breaks across a line
* en-dashes for numeric ranges (1778-1978, pp. 315-320)
* (no small caps: see smallcaps() below — the pass was removed 2026-09-15)
* p./pp., pl./pls. and chapter references become links to the original page
they name — note that the original's page numbers are the targets, not the
edition's, since `\origpage{N}` plants `page.N` where original page N begins
* \mbox round an inline specimen so it never breaks from adjacent punctuation
Never touched: comment lines, and the arguments of \erratum, \qslip, \ig,
\origpage, \fn, \includegraphics and the numeric arguments of \plateop/\pl/\tocl
— those quote the original exactly or are machine data.
"""
import re, glob, os, sys
SRC, OUT = "src/pages", "work/pages"
# Kept only as a record of what the removed pass covered. See smallcaps().
_FORMER_ACRONYMS = ["CSBC", "MSS", "SOAS", "BMS", "IOL", "IOR", "OUP", "SPG",
"EIC", "DCL", "FRS", "BEN", "MS", "BL"]
def build_maps():
plate, chap, subsec = {}, {}, {}
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
t = open(f).read()
for no, pg in re.findall(r"\\pl\{(\d+)\}\{.*?\}\{(\d+)\}", t):
plate[int(no)] = int(pg)
# \plx is the one List-of-Plates entry whose location is not a bare page
# number (plate 1 is "page 18", a plate facing the text), so its page
# reference is wrapped in \pg. Without this, plate 1 was the single plate
# whose in-text "pl. 1" references never linked.
for no, pg in re.findall(r"\\plx\{(\d+)\}\{.*?\}\{.*?\\pg\{(\d+)\}", t):
plate[int(no)] = int(pg)
for no, pg in re.findall(r"\\tocl\{[^}]*\}\{(\d+)\}\{.*?\}\{(\d+)\}", t):
chap[int(no)] = int(pg)
m = re.search(r"\\subchap\{(\d+)\.([ivx]+)", t)
o = re.search(r"\\origpage\{(\d+)\}", t)
if m and o: subsec[f"{m.group(1)}{m.group(2)}"] = int(o.group(1))
m2 = re.search(r"\\chapnum\{Chapter (\d+)\}", t)
if m2 and o: chap.setdefault(int(m2.group(1)), int(o.group(1)))
return plate, chap, subsec
PLATE, CHAP, SUBSEC = build_maps()
def link(target, shown):
# \xref, not \hyperlink: the preamble gives \xref a visible hairline rule,
# while \hyperlink stays unmarked for the whole-line Contents and plate entries.
return r"\xref{page.%d}{%s}" % (target, shown)
PROTECT = [
re.compile(r"\\(?:erratum|qslip)\{[^}]*\}(?:\{[^}]*\})?"),
re.compile(r"\\ig\{[^}]*\}"),
re.compile(r"\\includegraphics(?:\[[^\]]*\])?\{[^}]*\}"),
re.compile(r"\\(?:origpage|fn|hyperlink|hypertarget|pg|xref)\{[^}]*\}"),
re.compile(r"\\(?:plateop|plate)\{[^}]*\}\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:pl|tocl|toclnp|erratumline)(?:\{[^}]*\})?"),
# heading macros: their arguments become PDF bookmarks and running-head
# marks, so they must stay plain text — a link inside them breaks the build
re.compile(r"\\(?:chapnum|partstart|sectionstart)\{[^}]*\}\{[^}]*\}"),
re.compile(r"\\(?:chapstart|chaphead|subchap)(?:\[[^\]]*\])?\{[^}]*\}"),
]
def protect(text):
store = []
def keep(m):
store.append(m.group(0)); return "\x00%d\x00" % (len(store)-1)
for rx in PROTECT: text = rx.sub(keep, text)
return text, store
def restore(text, store):
return re.sub(r"\x00(\d+)\x00", lambda m: store[int(m.group(1))], text)
def xrefs(t):
# pl. 57 / pls. 83 and 84 — always this thesis's own plates, always linked
def plate_sub(m):
head, nums = m.group(1), m.group(2)
# An elided range end ("pls. 146-8" for 146-148, "142-5" for 142-145) is
# NOT a plate number of its own. Linking it as one sent "pls. 146-8" to
# plate 8 and "pls. 142-5" to plate 5 — three wrong links, found by the
# link audit 2026-09-15. Expand it from the leading digits of the number
# it is elided against before looking the plate up.
prev = None
def one(mm):
nonlocal prev
tok = mm.group(0); n = int(tok)
if prev is not None and n < prev:
full = int(str(prev)[:len(str(prev)) - len(tok)] + tok)
return link(PLATE[full], tok) if full in PLATE else tok
prev = n
return link(PLATE[n], tok) if n in PLATE else tok
return head + "~" + re.sub(r"\d+", one, nums)
t = re.sub(r"\b(pls?\.)\s+((?:\d+)(?:\s*(?:,|and|-|--)\s*\d+)*)", plate_sub, t)
# p. 63 / pp. 196-7 — link only where the thesis cites ITSELF. Ross's own
# pages are introduced by a self-reference ("see above, p. 63", "mentioned
# above, see p. 60", a bare "see p. 43"); a page in somebody else's book sits
# in a citation — after a title, an "ibid.", a shelfmark, or publication data.
# External markers are tested first, because a citation may also contain
# "see". Anything matching neither is left alone: a missing link is cheaper
# than one that lands on the wrong page.
external = (r"\\emph\{[^}]*\}[^.]{0,40}$", r"\bibid\b", r"\bop\. ?cit",
r"\([^()]*\b\d{4}\b[^()]*\)[^.]{0,30}$", r"\b(?:Records|MSS?|IOR|IOL|BMS|SPG|CSBC|BL)\b[^;]{0,45}$",
r"\b(?:vol|nos?|pt)\.\s*[\dIVXL][^.]{0,20}$", r"\b[IVXL]{1,5},\s*$",
r"\bedn\b", r"\brpt\b", r"\bfacing\b",
# a surname followed by a comma, then title/edition matter: the
# commonest shape of a citation in this thesis ("Halhed,
# \emph{Grammar}, p. xxiii", "See Halhed, Grammar, pp. 57")
r"[A-Z][a-zA-Z']+,[^;]{0,45}$")
internal = (r"\babove\b", r"\bbelow\b", r"\bsee\b", r"\bchapters?\b",
r"\bdiscussed\b", r"\bmentioned\b", r"\bcited\b", r"\bstated\b")
def page_sub(m):
w = t[max(0, m.start()-130):m.start()]
if any(re.search(rx, w) for rx in external): return m.group(0)
if not any(re.search(rx, w, re.I) for rx in internal): return m.group(0)
head, nums = m.group(1), m.group(2)
first = re.match(r"\d+", nums)
if not first: return m.group(0)
n = int(first.group(0))
if n > 431: return m.group(0) # beyond this thesis's last page
return head + "~" + link(n, first.group(0)) + nums[first.end():].replace("-", "--")
t = re.sub(r"\b(pp?\.)\s+(\d+(?:\s*-\s*\d+)?)", page_sub, t)
# chapter 5 / chapters 7 and 8 / chapter 3ii — always self-references
def chap_sub(m):
head, rest = m.group(1), m.group(2)
def one(mm):
tok = mm.group(0)
if tok in SUBSEC: return link(SUBSEC[tok], tok)
d = re.match(r"\d+", tok)
return link(CHAP[int(d.group(0))], tok) if d and tok.isdigit() and int(d.group(0)) in CHAP else tok
return head + "~" + re.sub(r"\d+(?:[ivx]+)?", one, rest)
t = re.sub(r"\b(chapters?)\s+(\d+(?:[ivx]+)?(?:\s*(?:,|and)\s*\d+(?:[ivx]+)?)*)",
chap_sub, t, flags=re.I)
return t
TIE_WORDS = r"(?:no|nos|vol|vols|fig|figs|Mr|Mrs|Dr|St|Revd|Rev|pt|Pt)\."
def ties(t):
t = re.sub(r"\b(%s)\s+(?=[\dIVXL])" % TIE_WORDS, r"\1~", t)
t = re.sub(r"\b(Part|Section|Book|Vol|Plate|Table|Figure)\s+(?=[\dIVXL])", r"\1~", t)
t = re.sub(r"\b(\d+)\s+(lbs?|pt|pts|dpi|mm|cm|in)\b", r"\1~\2", t)
t = re.sub(r"\b([A-Z][a-z]+)\s+(I{1,3}V?|IV|VI{0,3}|IX|XI{0,2})\b", r"\1~\2", t)
return t
def endashes(t):
return re.sub(r"(?<=\d)-(?=\d)", "--", t)
def smallcaps(t):
"""No-op since 2026-09-15 (Sammay): abbreviations stay plain uppercase.
The pass small-capped an explicit whitelist of institutional acronyms. That
whitelist could only ever be partial, and on the Abbreviations and
Conventions page the gap was plain: BFBS, LMS and MLCo sat in one column
against small-capped BL, BMS, EIC, IOL, OUP, SOAS and SPG. It also split a
shelfmark, setting a small-cap MS against a full-cap EUR in "MS EUR 30".
Completing the list was the obvious fix and is the wrong one. The original
is a typescript: a typewriter cannot set small caps, so the source defines
no abbreviation in them and every acronym on the page is plain uppercase.
Small caps were this edition's invention, and an inconsistent one. Rule 3 —
the original's typography stands — settles it."""
return t
def mbox_specimens(t):
return re.sub(r"(\\ig\{[^}]*\})", r"\\mbox{\1}", t)
def polish(text):
out_lines = []
for line in text.split("\n"):
if line.lstrip().startswith("%"):
out_lines.append(line); continue
body, store = protect(line)
body = xrefs(body)
body = ties(body)
body = endashes(body)
body = smallcaps(body)
body = restore(body, store)
body = mbox_specimens(body)
out_lines.append(body)
return "\n".join(out_lines)
if __name__ == "__main__":
os.makedirs(OUT, exist_ok=True)
n = 0
for f in sorted(glob.glob(f"{SRC}/p*.tex")):
open(os.path.join(OUT, os.path.basename(f)), "w").write(polish(open(f).read()))
n += 1
print(f"polished {n} pages -> {OUT}/ "
f"({len(PLATE)} plate targets, {len(CHAP)} chapters, {len(SUBSEC)} sections)")