Files
tepbc-ross-retyped/tools/epub_audit.py
T
bdeshiandClaude Sonnet 5.5 4424cc9887 Add the EPUB edition: converter, class-based audit, shared front matter
tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.

\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-09-29 14:36:18 +06:00

163 lines
6.9 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
r"""Audit the EPUB by CLASS of fault, not by instance. `make epub` runs it and
it exits non-zero on any failure.
Each check targets a class of fault the rendering sweep of 2026-09-26 found at
least one instance of. They are written against the class, so a new instance
of an old fault fails here even if it looks nothing like the first one:
residue any LaTeX syntax reaching the reader -- a backslash, a brace, or
a bracketed length. (First instance: "\\[0.6em]" printed as text.)
escaping generated markup re-escaped into visible text. (First: <div>,
<tr> showing as text after a second conversion pass.)
empty a block element holding nothing, or only page markers -- a stray
blank line. (First: 31 empty <p> around page breaks.)
links a fragment link resolving to no id in the book. (First: 199
Contents and Plates links pointing into the wrong file.)
images an <img> without width and height, or with a missing file.
xml any document that is not well-formed.
content any token of the transcription that does not reach the EPUB --
the class that holds silently dropped macros, dropped glyphs and
invisible page numbers alike. Runs the project's reprocheck.
typography the EPUB's typographic pass drifting from the PDF's: every range
en dash, tie, cross-reference link and unbreakable specimen that
tools/polish.py produces must reach the EPUB.
Empty table cells are deliberately NOT a fault: the Scheme of Transliteration's
last row is half-filled in the source itself.
"""
import glob
import html
import os
import posixpath
import re
import subprocess
import sys
import xml.dom.minidom as md
import zipfile
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
os.chdir(ROOT)
sys.path.insert(0, os.path.join(ROOT, 'tools'))
EPUB = sys.argv[1] if len(sys.argv) > 1 else 'ross-1988-retypeset.epub'
z = zipfile.ZipFile(EPUB)
X = {n: z.read(n).decode() for n in sorted(z.namelist()) if n.endswith('.xhtml')}
# every document a reader reads -- not the navigation, not the cover image page
CH = {n: d for n, d in X.items() if not re.search(r'/(nav|cover)\.xhtml$', n)}
def visible(d):
d = re.sub(r'<head>.*?</head>', '', d, flags=re.S)
return html.unescape(re.sub(r'<[^>]+>', ' ', d)) # tags -> space: cells never merge
def scan(pattern, docs, use_visible=True):
hits = []
for n, d in docs.items():
t = visible(d) if use_visible else d
for m in re.finditer(pattern, t):
ctx = t[max(0, m.start() - 40):m.end() + 40].replace('\n', ' ')
hits.append('%s: ...%s...' % (n.split('/')[-1], ctx))
return hits
results = []
def check(cls, what, hits):
results.append((cls, what, hits))
# ---- residue: LaTeX syntax in what the reader sees
check('residue', 'backslash', scan(r'\\', CH))
check('residue', 'brace', scan(r'[{}]', CH))
check('residue', 'bracketed length', scan(
r'\[\s*-?\d*\.?\d+\s*(em|ex|pt|pc|in|mm|cm|bp|sp)\s*\](\{[^}]*\})?', CH))
# ---- escaping: generated markup turned into text
check('escaping', 'escaped tag', scan(r'&lt;/?[a-zA-Z]', X, use_visible=False))
check('escaping', 'unrestored placeholder', scan(r'@@BLOCK\d+@@', X, use_visible=False))
# ---- empty: a block with nothing, or only page markers, in it
MARK = r'<span epub:type="pagebreak"[^>]*>[^<]*</span>'
check('empty', 'empty or marker-only block', scan(
r'<(p|div|blockquote|section|li|figcaption)(\s[^>]*)?>(\s|%s)*</\1>' % MARK,
CH, use_visible=False))
# ---- links
ids = {n: set(re.findall(r'\bid="([^"]+)"', d)) for n, d in X.items()}
bad = []
for n, d in X.items():
for h in re.findall(r'href="([^"]+)"', d):
if h.startswith(('http:', 'https:', 'mailto:')) or h.endswith('.css'):
continue
p, _, f = h.partition('#')
t = posixpath.normpath(posixpath.join(posixpath.dirname(n), p)) if p else n
if t not in X or (f and f not in ids[t]):
bad.append('%s -> %s' % (n.split('/')[-1], h))
check('links', 'fragment resolving nowhere', bad)
# ---- images
names = set(z.namelist())
check('images', 'img without width+height',
['%s: %s' % (n.split('/')[-1], t[:90]) for n, d in X.items()
for t in re.findall(r'<img [^>]*>', d) if not ('width=' in t and 'height=' in t)])
check('images', 'img file missing',
['%s: %s' % (n.split('/')[-1], s) for n, d in X.items()
for s in re.findall(r'src="([^"]+)"', d)
if posixpath.normpath(posixpath.join(posixpath.dirname(n), s)) not in names])
# ---- xml
mal = []
for n in z.namelist():
if n.endswith(('.xhtml', '.opf', '.xml', '.ncx')):
try:
md.parseString(z.read(n))
except Exception as e: # noqa: BLE001
mal.append('%s: %s' % (n, str(e)[:70]))
check('xml', 'not well-formed', mal)
# ---- content: every source token must reach the EPUB (the project's own gate)
r = subprocess.run([sys.executable, 'tools/reprocheck.py', EPUB],
capture_output=True, text=True)
first = r.stdout.splitlines()[0] if r.stdout else r.stderr.strip()
m = re.search(r'(\d+) tokens short', first)
check('content', 'tokens lost (reprocheck)',
[] if m and m.group(1) == '0' else [first] + r.stdout.splitlines()[1:6])
# ---- typography: the EPUB carries every transform polish.py makes
import polish # noqa: E402
# Expected: everything the PDF sets, the way build.sh sets it -- the pages
# through polish.py, the colophon and errata raw. (The colophon's ProQuest
# address carries an en dash of its own.)
pol = re.sub(r'(?m)(?<!\\)%.*$', '',
''.join(polish.polish(open(f).read())
for f in sorted(glob.glob('src/pages/p*.tex')))
+ open('src/colophon.tex').read() + open('src/errata.tex').read())
body = ''.join(CH.values())
text = ''.join(visible(d) for d in CH.values())
want = {
'range en dashes': (len(re.findall(r'(?<!-)--(?!-)', pol)), text.count('–')),
'cross-reference links': (pol.count('\\xref{'), body.count('class="xref"')),
'unbreakable specimens': (pol.count('\\mbox{'), body.count('class="nowrap"')),
}
drift = ['%s: polish.py makes %d, EPUB has %d' % (k, a, b)
for k, (a, b) in want.items() if a != b]
ties = pol.count('~')
if text.count(' ') < ties:
drift.append('ties: polish.py makes %d, EPUB has %d no-break spaces'
% (ties, text.count(' ')))
check('typography', 'drift from the PDF pass', drift)
# ---- report
w = max(len(c) + len(t) for c, t, _ in results) + 3
failed = 0
for cls, what, hits in results:
ok = not hits
failed += not ok
print('[%s] %-*s %4d%s' % (' ok ' if ok else 'FAIL', w, cls + ': ' + what, len(hits),
'' if ok else ' e.g. ' + hits[0][:150]))
print('\n%d failing check(s)' % failed)
sys.exit(1 if failed else 0)