Files
tepbc-ross-retyped/tools/reprocheck.py
T
bdeshiandClaude Sonnet 5.5 4424cc9887 Add the EPUB edition: converter, class-based audit, shared front matter
tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.

\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-09-29 14:36:18 +06:00

195 lines
8.1 KiB
Python

#!/usr/bin/env python3
r"""Reproduction check: every token in the transcription must reach the PDF.
The transcription in src/pages is the authority for what the edition should say;
the rendered PDF is what it does say. A token that appears N times in the source
and fewer than N times in the output is content the typesetter lost. Comparison
is by bag, not sequence, because endnotes and the Contents move text about.
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
python3 tools/reprocheck.py ross-1988-retypeset.epub
"""
import re, sys, glob, subprocess, unicodedata
from collections import Counter
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
# macros whose text we keep, by which argument(s) carry copy
KEEP = {
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
"fn": [1], # \fn{n}{text} - n is a mark
"plateop": [3], "plate": [2], # caption only
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
"mbox": [0], "textsuperscript": [0],
}
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
# environment -> (optional, mandatory) parameters that follow \begin{env}
ENV_PARAMS = {"addmargin": (1, 1), "tabular": (0, 1), "minipage": (1, 1)}
def argspans(s, i):
"""yield (start, end) of consecutive brace groups beginning at i"""
out = []
while i < len(s) and s[i] in " \t":
i += 1
while i < len(s) and s[i] == "{":
d, j = 0, i
while j < len(s):
if s[j] == "\\":
j += 2; continue
if s[j] == "{": d += 1
elif s[j] == "}":
d -= 1
if d == 0:
out.append((i + 1, j)); i = j + 1; break
j += 1
else:
break
return out, i
def detex(s):
s = re.sub(r"(?<!\\)%.*", "", s)
out, i = [], 0
while i < len(s):
c = s[i]
if c != "\\":
out.append(c); i += 1; continue
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
if not m:
i += 2
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
k = re.match(r"\s*\[[^\]]*\]", s[i:])
if k:
i += k.end()
out.append(" "); continue # \& \% \\ ...
name = m.group(1)
j = i + m.end()
while True: # skip [optional] args
k = s.find("[", j)
if k != -1 and s[j:k].strip() == "":
e = s.find("]", k)
if e == -1:
break
j = e + 1
else:
break
spans, after = argspans(s, j)
if name in ("begin", "end"): # env name and column spec are not copy
spans = []
# An environment's own parameters follow its name, after the
# {env} group -- \begin{addmargin}[0.47in]{0.5in}. They are layout,
# so skip exactly what each environment declares; left in, they
# were counted as copy ("0", "47", "in") that no edition prints.
env = re.match(r"\s*\{(\w+)\}", s[j:])
if name == "begin" and env and env.group(1) in ENV_PARAMS:
after = j + env.end()
nopt, nman = ENV_PARAMS[env.group(1)]
for _ in range(nopt):
k = re.match(r"\s*\[[^\]]*\]", s[after:])
if k:
after += k.end()
if nman:
_sp, after = argspans(s, after)
if len(_sp) > nman: # keep only the declared count
after = _sp[nman - 1][1] + 1
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
k = re.match(r"\s*[0-9]+", s[after:])
if k:
after += k.end()
if name in DROP:
out.append(" "); i = after; continue
keep = KEEP.get(name)
if keep is None:
out.append(" ")
for a, b in spans:
out.append(detex(s[a:b]) + " ")
else:
for k in keep:
if k < len(spans):
a, b = spans[k]
out.append(" " + detex(s[a:b]) + " ")
i = after
return "".join(out)
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
def tokens(text):
text = unicodedata.normalize("NFC", text)
for k, v in LIG.items():
text = text.replace(k, v)
text = text.replace("\u2019", "'").replace("\u2018", "'")
text = re.sub(r"[\u2010-\u2015]", "-", text)
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
src = []
files = sorted(glob.glob("src/pages/p*.tex"))
for f in files:
src.append(detex(open(f, encoding="utf-8").read()))
A = tokens("\n".join(src))
def epub_text(path):
"""Visible text of the EPUB's chapter documents. Tags become spaces, so
table cells and adjacent elements never merge into one token. The
navigation document is left out on purpose: it repeats every heading, and
counting it would hide a heading lost from the chapter itself."""
import zipfile, html as _html
z = zipfile.ZipFile(path)
parts = []
for n in sorted(z.namelist()):
# the chapters, and the one notes document the endnotes now live in;
# not the colophon or errata, which are not transcription
if re.search(r"/(ch\d+|notes)\.xhtml$", n):
d = re.sub(r"<head>.*?</head>", "", z.read(n).decode(), flags=re.S)
parts.append(_html.unescape(re.sub(r"<[^>]+>", " ", d)))
return "\n".join(parts)
# The same check serves both editions: every token of the transcription must
# reach the output, whichever output it is.
if PDF.endswith(".epub"):
pdftxt = epub_text(PDF)
else:
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
capture_output=True, text=True).stdout
B = tokens(pdftxt)
where = {}
for f, txt in zip(files, src):
for t in set(tokens(txt)):
where.setdefault(t, []).append(f[-8:-4])
# pdftotext merges two tokens wherever the gap is too small to register as a
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
# the extractor, not in the PDF, and they show up as a token that survives only
# inside a longer one - so report them apart instead of burying the real ones.
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
merged, real = [], []
for t in A:
if B[t] >= A[t]:
continue
row = (A[t] - B[t], t, A[t], B[t])
probe = unicodedata.normalize("NFD", t)
(merged if stream.count(probe) >= A[t] else real).append(row)
real.sort(key=lambda r: (-r[0], r[1]))
merged.sort(key=lambda r: (-r[0], r[1]))
deficits = real
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
f"({len(merged)} more merged into a neighbour by the extractor)")
for n, t, a, b in deficits:
pp = where.get(t, [])
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
if merged:
print("extractor merges (not losses):",
", ".join(f"{t}(-{n})" for n, t, a, b in merged))