#!/usr/bin/env python3 r"""Reproduction check: every token in the transcription must reach the PDF. The transcription in src/pages is the authority for what the edition should say; the rendered PDF is what it does say. A token that appears N times in the source and fewer than N times in the output is content the typesetter lost. Comparison is by bag, not sequence, because endnotes and the Contents move text about. Found the biblist \hangafter=1 digit-swallowing bug (2026-09-14). python3 tools/reprocheck.py [ross-1988-retypeset.pdf] """ import re, sys, glob, subprocess, unicodedata from collections import Counter PDF = sys.argv[1] if len(sys.argv) > 1 else "ross-1988-retypeset.pdf" # macros whose text we keep, by which argument(s) carry copy KEEP = { "emph": [0], "textbf": [0], "textit": [0], "textsc": [0], "underline": [0], "unsure": [0], "qslip": [0], "erratum": [0], # \erratum{corrected}{as printed} "fn": [1], # \fn{n}{text} - n is a mark "plateop": [3], "plate": [2], # caption only "chapstart": [0], "chaphead": [0], "subhead": [0], "bibgroup": [0], "bibhead": [0], "partstart": [0, 1], "sectionstart": [0, 1], "matterstart": [0], "chapnum": [0, 1], "pl": [1], "tocl": [2], "toclnp": [2], "pg": [0], "origpage": [0], "mbox": [0], "textsuperscript": [0], } DROP = {"ig", "fig", "includegraphics", "label", "index", "vspace", "hspace", "setlength", "pdfbookmark", "addcontentsline", "input", "hangindent", "thispagestyle", "pagestyle", "setstretch", "addmargin", "addvspace", "vskip", "hskip", "rule", "makebox", "raisebox", "resizebox", "XeTeXglyph"} def argspans(s, i): """yield (start, end) of consecutive brace groups beginning at i""" out = [] while i < len(s) and s[i] in " \t": i += 1 while i < len(s) and s[i] == "{": d, j = 0, i while j < len(s): if s[j] == "\\": j += 2; continue if s[j] == "{": d += 1 elif s[j] == "}": d -= 1 if d == 0: out.append((i + 1, j)); i = j + 1; break j += 1 else: break return out, i def detex(s): s = re.sub(r"(? "em55"), and # the two Bengali forms of a Scheme-of-Transliteration cell. Those are losses in # the extractor, not in the PDF, and they show up as a token that survives only # inside a longer one - so report them apart instead of burying the real ones. # NFD so a precomposed nukta form (U+09DC) still shows its base letter stream = unicodedata.normalize("NFD", re.sub(r"\s+", " ", pdftxt.lower())) merged, real = [], [] for t in A: if B[t] >= A[t]: continue row = (A[t] - B[t], t, A[t], B[t]) probe = unicodedata.normalize("NFD", t) (merged if stream.count(probe) >= A[t] else real).append(row) real.sort(key=lambda r: (-r[0], r[1])) merged.sort(key=lambda r: (-r[0], r[1])) deficits = real print(f"{len(files)} page files, {sum(A.values())} source tokens, " f"{sum(B.values())} pdf tokens, {len(real)} tokens short " f"({len(merged)} more merged into a neighbour by the extractor)") for n, t, a, b in deficits: pp = where.get(t, []) tail = " ".join(pp) if len(pp) <= 8 else " ".join(pp[:8]) + " ..." print(f" -{n:<4} {t!r:24} src {a:<4} pdf {b:<4} p{tail}") if merged: print("extractor merges (not losses):", ", ".join(f"{t}(-{n})" for n, t, a, b in merged))