Files
tepbc-ross-retyped/tools/reprocheck.py
T
bdeshiandClaude Opus 5 82849eb94d Initial commit: Ross 1988 re-typeset edition
A searchable, re-typeset edition of Fiona Ross, "The Evolution of the
Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988),
transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178
plates and 410 inline type specimens cut from the scan; 51 errata.

Tracked: the transcription (src/pages), the preamble and its typographic
decisions, the cut images (plates/ — not reliably regenerable, the crop
specs for the inline cuts were never scripted), tools, and the four
working documents.

Not tracked: the built PDF, which `make` remakes from src/ and plates/;
the ProQuest scan under source/, which is third-party and needed only by
`make prep` and `make plate`; scans/ and work/, both regenerable.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 19:03:25 +06:00

154 lines
6.0 KiB
Python

#!/usr/bin/env python3
r"""Reproduction check: every token in the transcription must reach the PDF.
The transcription in src/pages is the authority for what the edition should say;
the rendered PDF is what it does say. A token that appears N times in the source
and fewer than N times in the output is content the typesetter lost. Comparison
is by bag, not sequence, because endnotes and the Contents move text about.
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
"""
import re, sys, glob, subprocess, unicodedata
from collections import Counter
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
# macros whose text we keep, by which argument(s) carry copy
KEEP = {
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
"fn": [1], # \fn{n}{text} - n is a mark
"plateop": [3], "plate": [2], # caption only
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
"mbox": [0], "textsuperscript": [0],
}
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
def argspans(s, i):
"""yield (start, end) of consecutive brace groups beginning at i"""
out = []
while i < len(s) and s[i] in " \t":
i += 1
while i < len(s) and s[i] == "{":
d, j = 0, i
while j < len(s):
if s[j] == "\\":
j += 2; continue
if s[j] == "{": d += 1
elif s[j] == "}":
d -= 1
if d == 0:
out.append((i + 1, j)); i = j + 1; break
j += 1
else:
break
return out, i
def detex(s):
s = re.sub(r"(?<!\\)%.*", "", s)
out, i = [], 0
while i < len(s):
c = s[i]
if c != "\\":
out.append(c); i += 1; continue
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
if not m:
i += 2
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
k = re.match(r"\s*\[[^\]]*\]", s[i:])
if k:
i += k.end()
out.append(" "); continue # \& \% \\ ...
name = m.group(1)
j = i + m.end()
while True: # skip [optional] args
k = s.find("[", j)
if k != -1 and s[j:k].strip() == "":
e = s.find("]", k)
if e == -1:
break
j = e + 1
else:
break
spans, after = argspans(s, j)
if name in ("begin", "end"): # env name and column spec are not copy
spans = []
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
k = re.match(r"\s*[0-9]+", s[after:])
if k:
after += k.end()
if name in DROP:
out.append(" "); i = after; continue
keep = KEEP.get(name)
if keep is None:
out.append(" ")
for a, b in spans:
out.append(detex(s[a:b]) + " ")
else:
for k in keep:
if k < len(spans):
a, b = spans[k]
out.append(" " + detex(s[a:b]) + " ")
i = after
return "".join(out)
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
def tokens(text):
text = unicodedata.normalize("NFC", text)
for k, v in LIG.items():
text = text.replace(k, v)
text = text.replace("\u2019", "'").replace("\u2018", "'")
text = re.sub(r"[\u2010-\u2015]", "-", text)
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
src = []
files = sorted(glob.glob("src/pages/p*.tex"))
for f in files:
src.append(detex(open(f, encoding="utf-8").read()))
A = tokens("\n".join(src))
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
capture_output=True, text=True).stdout
B = tokens(pdftxt)
where = {}
for f, txt in zip(files, src):
for t in set(tokens(txt)):
where.setdefault(t, []).append(f[-8:-4])
# pdftotext merges two tokens wherever the gap is too small to register as a
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
# the extractor, not in the PDF, and they show up as a token that survives only
# inside a longer one - so report them apart instead of burying the real ones.
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
merged, real = [], []
for t in A:
if B[t] >= A[t]:
continue
row = (A[t] - B[t], t, A[t], B[t])
probe = unicodedata.normalize("NFD", t)
(merged if stream.count(probe) >= A[t] else real).append(row)
real.sort(key=lambda r: (-r[0], r[1]))
merged.sort(key=lambda r: (-r[0], r[1]))
deficits = real
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
f"({len(merged)} more merged into a neighbour by the extractor)")
for n, t, a, b in deficits:
pp = where.get(t, [])
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
if merged:
print("extractor merges (not losses):",
", ".join(f"{t}(-{n})" for n, t, a, b in merged))