A searchable, re-typeset edition of Fiona Ross, "The Evolution of the Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988), transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178 plates and 410 inline type specimens cut from the scan; 51 errata. Tracked: the transcription (src/pages), the preamble and its typographic decisions, the cut images (plates/ — not reliably regenerable, the crop specs for the inline cuts were never scripted), tools, and the four working documents. Not tracked: the built PDF, which `make` remakes from src/ and plates/; the ProQuest scan under source/, which is third-party and needed only by `make prep` and `make plate`; scans/ and work/, both regenerable. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
154 lines
6.0 KiB
Python
154 lines
6.0 KiB
Python
#!/usr/bin/env python3
|
|
r"""Reproduction check: every token in the transcription must reach the PDF.
|
|
|
|
The transcription in src/pages is the authority for what the edition should say;
|
|
the rendered PDF is what it does say. A token that appears N times in the source
|
|
and fewer than N times in the output is content the typesetter lost. Comparison
|
|
is by bag, not sequence, because endnotes and the Contents move text about.
|
|
|
|
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
|
|
|
|
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
|
|
"""
|
|
import re, sys, glob, subprocess, unicodedata
|
|
from collections import Counter
|
|
|
|
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
|
|
|
|
# macros whose text we keep, by which argument(s) carry copy
|
|
KEEP = {
|
|
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
|
|
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
|
|
"fn": [1], # \fn{n}{text} - n is a mark
|
|
"plateop": [3], "plate": [2], # caption only
|
|
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
|
|
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
|
|
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
|
|
"mbox": [0], "textsuperscript": [0],
|
|
}
|
|
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
|
|
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
|
|
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
|
|
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
|
|
|
|
def argspans(s, i):
|
|
"""yield (start, end) of consecutive brace groups beginning at i"""
|
|
out = []
|
|
while i < len(s) and s[i] in " \t":
|
|
i += 1
|
|
while i < len(s) and s[i] == "{":
|
|
d, j = 0, i
|
|
while j < len(s):
|
|
if s[j] == "\\":
|
|
j += 2; continue
|
|
if s[j] == "{": d += 1
|
|
elif s[j] == "}":
|
|
d -= 1
|
|
if d == 0:
|
|
out.append((i + 1, j)); i = j + 1; break
|
|
j += 1
|
|
else:
|
|
break
|
|
return out, i
|
|
|
|
def detex(s):
|
|
s = re.sub(r"(?<!\\)%.*", "", s)
|
|
out, i = [], 0
|
|
while i < len(s):
|
|
c = s[i]
|
|
if c != "\\":
|
|
out.append(c); i += 1; continue
|
|
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
|
|
if not m:
|
|
i += 2
|
|
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
|
|
k = re.match(r"\s*\[[^\]]*\]", s[i:])
|
|
if k:
|
|
i += k.end()
|
|
out.append(" "); continue # \& \% \\ ...
|
|
name = m.group(1)
|
|
j = i + m.end()
|
|
while True: # skip [optional] args
|
|
k = s.find("[", j)
|
|
if k != -1 and s[j:k].strip() == "":
|
|
e = s.find("]", k)
|
|
if e == -1:
|
|
break
|
|
j = e + 1
|
|
else:
|
|
break
|
|
spans, after = argspans(s, j)
|
|
if name in ("begin", "end"): # env name and column spec are not copy
|
|
spans = []
|
|
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
|
|
k = re.match(r"\s*[0-9]+", s[after:])
|
|
if k:
|
|
after += k.end()
|
|
if name in DROP:
|
|
out.append(" "); i = after; continue
|
|
keep = KEEP.get(name)
|
|
if keep is None:
|
|
out.append(" ")
|
|
for a, b in spans:
|
|
out.append(detex(s[a:b]) + " ")
|
|
else:
|
|
for k in keep:
|
|
if k < len(spans):
|
|
a, b = spans[k]
|
|
out.append(" " + detex(s[a:b]) + " ")
|
|
i = after
|
|
return "".join(out)
|
|
|
|
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
|
|
|
|
def tokens(text):
|
|
text = unicodedata.normalize("NFC", text)
|
|
for k, v in LIG.items():
|
|
text = text.replace(k, v)
|
|
text = text.replace("\u2019", "'").replace("\u2018", "'")
|
|
text = re.sub(r"[\u2010-\u2015]", "-", text)
|
|
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
|
|
|
|
src = []
|
|
files = sorted(glob.glob("src/pages/p*.tex"))
|
|
for f in files:
|
|
src.append(detex(open(f, encoding="utf-8").read()))
|
|
A = tokens("\n".join(src))
|
|
|
|
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
|
|
capture_output=True, text=True).stdout
|
|
B = tokens(pdftxt)
|
|
|
|
where = {}
|
|
for f, txt in zip(files, src):
|
|
for t in set(tokens(txt)):
|
|
where.setdefault(t, []).append(f[-8:-4])
|
|
|
|
# pdftotext merges two tokens wherever the gap is too small to register as a
|
|
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
|
|
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
|
|
# the extractor, not in the PDF, and they show up as a token that survives only
|
|
# inside a longer one - so report them apart instead of burying the real ones.
|
|
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
|
|
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
|
|
merged, real = [], []
|
|
for t in A:
|
|
if B[t] >= A[t]:
|
|
continue
|
|
row = (A[t] - B[t], t, A[t], B[t])
|
|
probe = unicodedata.normalize("NFD", t)
|
|
(merged if stream.count(probe) >= A[t] else real).append(row)
|
|
real.sort(key=lambda r: (-r[0], r[1]))
|
|
merged.sort(key=lambda r: (-r[0], r[1]))
|
|
deficits = real
|
|
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
|
|
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
|
|
f"({len(merged)} more merged into a neighbour by the extractor)")
|
|
for n, t, a, b in deficits:
|
|
pp = where.get(t, [])
|
|
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
|
|
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
|
|
if merged:
|
|
print("extractor merges (not losses):",
|
|
", ".join(f"{t}(-{n})" for n, t, a, b in merged))
|