tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.
\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
195 lines
8.1 KiB
Python
195 lines
8.1 KiB
Python
#!/usr/bin/env python3
|
|
r"""Reproduction check: every token in the transcription must reach the PDF.
|
|
|
|
The transcription in src/pages is the authority for what the edition should say;
|
|
the rendered PDF is what it does say. A token that appears N times in the source
|
|
and fewer than N times in the output is content the typesetter lost. Comparison
|
|
is by bag, not sequence, because endnotes and the Contents move text about.
|
|
|
|
Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14).
|
|
|
|
python3 tools/reprocheck.py [ross-1988-retypeset.pdf]
|
|
python3 tools/reprocheck.py ross-1988-retypeset.epub
|
|
"""
|
|
import re, sys, glob, subprocess, unicodedata
|
|
from collections import Counter
|
|
|
|
PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf"
|
|
|
|
# macros whose text we keep, by which argument(s) carry copy
|
|
KEEP = {
|
|
"emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0],
|
|
"unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed}
|
|
"fn": [1], # \fn{n}{text} - n is a mark
|
|
"plateop": [3], "plate": [2], # caption only
|
|
"chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0],
|
|
"partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1],
|
|
"pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0],
|
|
"mbox": [0], "textsuperscript": [0],
|
|
}
|
|
DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace",
|
|
"setlength", "pdfbookmark", "addcontentsline", "input", "hangindent",
|
|
"thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace",
|
|
"vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"}
|
|
|
|
# environment -> (optional, mandatory) parameters that follow \begin{env}
|
|
ENV_PARAMS = {"addmargin": (1, 1), "tabular": (0, 1), "minipage": (1, 1)}
|
|
|
|
def argspans(s, i):
|
|
"""yield (start, end) of consecutive brace groups beginning at i"""
|
|
out = []
|
|
while i < len(s) and s[i] in " \t":
|
|
i += 1
|
|
while i < len(s) and s[i] == "{":
|
|
d, j = 0, i
|
|
while j < len(s):
|
|
if s[j] == "\\":
|
|
j += 2; continue
|
|
if s[j] == "{": d += 1
|
|
elif s[j] == "}":
|
|
d -= 1
|
|
if d == 0:
|
|
out.append((i + 1, j)); i = j + 1; break
|
|
j += 1
|
|
else:
|
|
break
|
|
return out, i
|
|
|
|
def detex(s):
|
|
s = re.sub(r"(?<!\\)%.*", "", s)
|
|
out, i = [], 0
|
|
while i < len(s):
|
|
c = s[i]
|
|
if c != "\\":
|
|
out.append(c); i += 1; continue
|
|
m = re.match(r"\\([A-Za-z@]+)\*?", s[i:])
|
|
if not m:
|
|
i += 2
|
|
if s[i - 2:i] == "\\\\": # \\[0.6em] - the skip is a length
|
|
k = re.match(r"\s*\[[^\]]*\]", s[i:])
|
|
if k:
|
|
i += k.end()
|
|
out.append(" "); continue # \& \% \\ ...
|
|
name = m.group(1)
|
|
j = i + m.end()
|
|
while True: # skip [optional] args
|
|
k = s.find("[", j)
|
|
if k != -1 and s[j:k].strip() == "":
|
|
e = s.find("]", k)
|
|
if e == -1:
|
|
break
|
|
j = e + 1
|
|
else:
|
|
break
|
|
spans, after = argspans(s, j)
|
|
if name in ("begin", "end"): # env name and column spec are not copy
|
|
spans = []
|
|
# An environment's own parameters follow its name, after the
|
|
# {env} group -- \begin{addmargin}[0.47in]{0.5in}. They are layout,
|
|
# so skip exactly what each environment declares; left in, they
|
|
# were counted as copy ("0", "47", "in") that no edition prints.
|
|
env = re.match(r"\s*\{(\w+)\}", s[j:])
|
|
if name == "begin" and env and env.group(1) in ENV_PARAMS:
|
|
after = j + env.end()
|
|
nopt, nman = ENV_PARAMS[env.group(1)]
|
|
for _ in range(nopt):
|
|
k = re.match(r"\s*\[[^\]]*\]", s[after:])
|
|
if k:
|
|
after += k.end()
|
|
if nman:
|
|
_sp, after = argspans(s, after)
|
|
if len(_sp) > nman: # keep only the declared count
|
|
after = _sp[nman - 1][1] + 1
|
|
if name == "XeTeXglyph": # \XeTeXglyph 824 - unbraced slot id
|
|
k = re.match(r"\s*[0-9]+", s[after:])
|
|
if k:
|
|
after += k.end()
|
|
if name in DROP:
|
|
out.append(" "); i = after; continue
|
|
keep = KEEP.get(name)
|
|
if keep is None:
|
|
out.append(" ")
|
|
for a, b in spans:
|
|
out.append(detex(s[a:b]) + " ")
|
|
else:
|
|
for k in keep:
|
|
if k < len(spans):
|
|
a, b = spans[k]
|
|
out.append(" " + detex(s[a:b]) + " ")
|
|
i = after
|
|
return "".join(out)
|
|
|
|
LIG = {"\ufb00": "ff", "\ufb01": "fi", "\ufb02": "fl", "\ufb03": "ffi", "\ufb04": "ffl"}
|
|
|
|
def tokens(text):
|
|
text = unicodedata.normalize("NFC", text)
|
|
for k, v in LIG.items():
|
|
text = text.replace(k, v)
|
|
text = text.replace("\u2019", "'").replace("\u2018", "'")
|
|
text = re.sub(r"[\u2010-\u2015]", "-", text)
|
|
return Counter(t.lower() for t in re.findall(r"[0-9]+|[^\W\d_]+", text, re.UNICODE))
|
|
|
|
src = []
|
|
files = sorted(glob.glob("src/pages/p*.tex"))
|
|
for f in files:
|
|
src.append(detex(open(f, encoding="utf-8").read()))
|
|
A = tokens("\n".join(src))
|
|
|
|
def epub_text(path):
|
|
"""Visible text of the EPUB's chapter documents. Tags become spaces, so
|
|
table cells and adjacent elements never merge into one token. The
|
|
navigation document is left out on purpose: it repeats every heading, and
|
|
counting it would hide a heading lost from the chapter itself."""
|
|
import zipfile, html as _html
|
|
z = zipfile.ZipFile(path)
|
|
parts = []
|
|
for n in sorted(z.namelist()):
|
|
# the chapters, and the one notes document the endnotes now live in;
|
|
# not the colophon or errata, which are not transcription
|
|
if re.search(r"/(ch\d+|notes)\.xhtml$", n):
|
|
d = re.sub(r"<head>.*?</head>", "", z.read(n).decode(), flags=re.S)
|
|
parts.append(_html.unescape(re.sub(r"<[^>]+>", " ", d)))
|
|
return "\n".join(parts)
|
|
|
|
# The same check serves both editions: every token of the transcription must
|
|
# reach the output, whichever output it is.
|
|
if PDF.endswith(".epub"):
|
|
pdftxt = epub_text(PDF)
|
|
else:
|
|
pdftxt = subprocess.run(["pdftotext", "-layout", PDF, "-"],
|
|
capture_output=True, text=True).stdout
|
|
B = tokens(pdftxt)
|
|
|
|
where = {}
|
|
for f, txt in zip(files, src):
|
|
for t in set(tokens(txt)):
|
|
where.setdefault(t, []).append(f[-8:-4])
|
|
|
|
# pdftotext merges two tokens wherever the gap is too small to register as a
|
|
# space: an endnote superscript onto the word before it (em.55 -> "em55"), and
|
|
# the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in
|
|
# the extractor, not in the PDF, and they show up as a token that survives only
|
|
# inside a longer one - so report them apart instead of burying the real ones.
|
|
# NFD so a precomposed nukta form (U+09DC) still shows its base letter
|
|
stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower()))
|
|
merged, real = [], []
|
|
for t in A:
|
|
if B[t] >= A[t]:
|
|
continue
|
|
row = (A[t] - B[t], t, A[t], B[t])
|
|
probe = unicodedata.normalize("NFD", t)
|
|
(merged if stream.count(probe) >= A[t] else real).append(row)
|
|
real.sort(key=lambda r: (-r[0], r[1]))
|
|
merged.sort(key=lambda r: (-r[0], r[1]))
|
|
deficits = real
|
|
print(f"{len(files)} page files, {sum(A.values())} source tokens, "
|
|
f"{sum(B.values())} pdf tokens, {len(real)} tokens short "
|
|
f"({len(merged)} more merged into a neighbour by the extractor)")
|
|
for n, t, a, b in deficits:
|
|
pp = where.get(t, [])
|
|
tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..."
|
|
print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}")
|
|
if merged:
|
|
print("extractor merges (not losses):",
|
|
", ".join(f"{t}(-{n})" for n, t, a, b in merged))
|