Initial commit: Ross 1988 re-typeset edition
A searchable, re-typeset edition of Fiona Ross, "The Evolution of the Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988), transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178 plates and 410 inline type specimens cut from the scan; 51 errata. Tracked: the transcription (src/pages), the preamble and its typographic decisions, the cut images (plates/ — not reliably regenerable, the crop specs for the inline cuts were never scripted), tools, and the four working documents. Not tracked: the built PDF, which `make` remakes from src/ and plates/; the ProQuest scan under source/, which is third-party and needed only by `make prep` and `make plate`; scans/ and work/, both regenerable. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env python3
|
||||
r"""Reproduction check: every token in the transcription must reach the PDF.
|
||||
|
||||
The transcription in src/pages is the authority for what the edition should say;
|
||||
the rendered PDF is what it does say. A token that appears N times in the source
|
||||
and fewer than N times in the output is content the typesetter lost. Comparison
|
||||
is by bag, not sequence, because endnotes and the Contents move text about.
|
||||
|
||||
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
|
||||
|
||||
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
|
||||
"""
|
||||
import re, sys, glob, subprocess, unicodedata
|
||||
from collections import Counter
|
||||
|
||||
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
|
||||
|
||||
# macros whose text we keep, by which argument(s) carry copy
|
||||
KEEP = {
|
||||
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
|
||||
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
|
||||
"fn": [1], # \fn{n}{text} - n is a mark
|
||||
"plateop": [3], "plate": [2], # caption only
|
||||
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
|
||||
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
|
||||
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
|
||||
"mbox": [0], "textsuperscript": [0],
|
||||
}
|
||||
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
|
||||
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
|
||||
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
|
||||
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
|
||||
|
||||
def argspans(s, i):
|
||||
"""yield (start, end) of consecutive brace groups beginning at i"""
|
||||
out = []
|
||||
while i < len(s) and s[i] in " \t":
|
||||
i += 1
|
||||
while i < len(s) and s[i] == "{":
|
||||
d, j = 0, i
|
||||
while j < len(s):
|
||||
if s[j] == "\\":
|
||||
j += 2; continue
|
||||
if s[j] == "{": d += 1
|
||||
elif s[j] == "}":
|
||||
d -= 1
|
||||
if d == 0:
|
||||
out.append((i + 1, j)); i = j + 1; break
|
||||
j += 1
|
||||
else:
|
||||
break
|
||||
return out, i
|
||||
|
||||
def detex(s):
|
||||
s = re.sub(r"(?<!\\)%.*", "", s)
|
||||
out, i = [], 0
|
||||
while i < len(s):
|
||||
c = s[i]
|
||||
if c != "\\":
|
||||
out.append(c); i += 1; continue
|
||||
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
|
||||
if not m:
|
||||
i += 2
|
||||
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
|
||||
k = re.match(r"\s*\[[^\]]*\]", s[i:])
|
||||
if k:
|
||||
i += k.end()
|
||||
out.append(" "); continue # \& \% \\ ...
|
||||
name = m.group(1)
|
||||
j = i + m.end()
|
||||
while True: # skip [optional] args
|
||||
k = s.find("[", j)
|
||||
if k != -1 and s[j:k].strip() == "":
|
||||
e = s.find("]", k)
|
||||
if e == -1:
|
||||
break
|
||||
j = e + 1
|
||||
else:
|
||||
break
|
||||
spans, after = argspans(s, j)
|
||||
if name in ("begin", "end"): # env name and column spec are not copy
|
||||
spans = []
|
||||
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
|
||||
k = re.match(r"\s*[0-9]+", s[after:])
|
||||
if k:
|
||||
after += k.end()
|
||||
if name in DROP:
|
||||
out.append(" "); i = after; continue
|
||||
keep = KEEP.get(name)
|
||||
if keep is None:
|
||||
out.append(" ")
|
||||
for a, b in spans:
|
||||
out.append(detex(s[a:b]) + " ")
|
||||
else:
|
||||
for k in keep:
|
||||
if k < len(spans):
|
||||
a, b = spans[k]
|
||||
out.append(" " + detex(s[a:b]) + " ")
|
||||
i = after
|
||||
return "".join(out)
|
||||
|
||||
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
|
||||
|
||||
def tokens(text):
|
||||
text = unicodedata.normalize("NFC", text)
|
||||
for k, v in LIG.items():
|
||||
text = text.replace(k, v)
|
||||
text = text.replace("\u2019", "'").replace("\u2018", "'")
|
||||
text = re.sub(r"[\u2010-\u2015]", "-", text)
|
||||
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
|
||||
|
||||
src = []
|
||||
files = sorted(glob.glob("src/pages/p*.tex"))
|
||||
for f in files:
|
||||
src.append(detex(open(f, encoding="utf-8").read()))
|
||||
A = tokens("\n".join(src))
|
||||
|
||||
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
|
||||
capture_output=True, text=True).stdout
|
||||
B = tokens(pdftxt)
|
||||
|
||||
where = {}
|
||||
for f, txt in zip(files, src):
|
||||
for t in set(tokens(txt)):
|
||||
where.setdefault(t, []).append(f[-8:-4])
|
||||
|
||||
# pdftotext merges two tokens wherever the gap is too small to register as a
|
||||
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
|
||||
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
|
||||
# the extractor, not in the PDF, and they show up as a token that survives only
|
||||
# inside a longer one - so report them apart instead of burying the real ones.
|
||||
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
|
||||
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
|
||||
merged, real = [], []
|
||||
for t in A:
|
||||
if B[t] >= A[t]:
|
||||
continue
|
||||
row = (A[t] - B[t], t, A[t], B[t])
|
||||
probe = unicodedata.normalize("NFD", t)
|
||||
(merged if stream.count(probe) >= A[t] else real).append(row)
|
||||
real.sort(key=lambda r: (-r[0], r[1]))
|
||||
merged.sort(key=lambda r: (-r[0], r[1]))
|
||||
deficits = real
|
||||
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
|
||||
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
|
||||
f"({len(merged)} more merged into a neighbour by the extractor)")
|
||||
for n, t, a, b in deficits:
|
||||
pp = where.get(t, [])
|
||||
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
|
||||
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
|
||||
if merged:
|
||||
print("extractor merges (not losses):",
|
||||
", ".join(f"{t}(-{n})" for n, t, a, b in merged))
|
||||
Reference in New Issue
Block a user