tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.
\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
163 lines
6.9 KiB
Python
163 lines
6.9 KiB
Python
#!/usr/bin/env python3
|
||
r"""Audit the EPUB by CLASS of fault, not by instance. `make epub` runs it and
|
||
it exits non-zero on any failure.
|
||
|
||
Each check targets a class of fault the rendering sweep of 2026-09-26 found at
|
||
least one instance of. They are written against the class, so a new instance
|
||
of an old fault fails here even if it looks nothing like the first one:
|
||
|
||
residue any LaTeX syntax reaching the reader -- a backslash, a brace, or
|
||
a bracketed length. (First instance: "\\[0.6em]" printed as text.)
|
||
escaping generated markup re-escaped into visible text. (First: <div>,
|
||
<tr> showing as text after a second conversion pass.)
|
||
empty a block element holding nothing, or only page markers -- a stray
|
||
blank line. (First: 31 empty <p> around page breaks.)
|
||
links a fragment link resolving to no id in the book. (First: 199
|
||
Contents and Plates links pointing into the wrong file.)
|
||
images an <img> without width and height, or with a missing file.
|
||
xml any document that is not well-formed.
|
||
content any token of the transcription that does not reach the EPUB --
|
||
the class that holds silently dropped macros, dropped glyphs and
|
||
invisible page numbers alike. Runs the project's reprocheck.
|
||
typography the EPUB's typographic pass drifting from the PDF's: every range
|
||
en dash, tie, cross-reference link and unbreakable specimen that
|
||
tools/polish.py produces must reach the EPUB.
|
||
|
||
Empty table cells are deliberately NOT a fault: the Scheme of Transliteration's
|
||
last row is half-filled in the source itself.
|
||
"""
|
||
import glob
|
||
import html
|
||
import os
|
||
import posixpath
|
||
import re
|
||
import subprocess
|
||
import sys
|
||
import xml.dom.minidom as md
|
||
import zipfile
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
os.chdir(ROOT)
|
||
sys.path.insert(0, os.path.join(ROOT, 'tools'))
|
||
EPUB = sys.argv[1] if len(sys.argv) > 1 else 'ross-1988-retypeset.epub'
|
||
|
||
z = zipfile.ZipFile(EPUB)
|
||
X = {n: z.read(n).decode() for n in sorted(z.namelist()) if n.endswith('.xhtml')}
|
||
# every document a reader reads -- not the navigation, not the cover image page
|
||
CH = {n: d for n, d in X.items() if not re.search(r'/(nav|cover)\.xhtml$', n)}
|
||
|
||
|
||
def visible(d):
|
||
d = re.sub(r'<head>.*?</head>', '', d, flags=re.S)
|
||
return html.unescape(re.sub(r'<[^>]+>', ' ', d)) # tags -> space: cells never merge
|
||
|
||
|
||
def scan(pattern, docs, use_visible=True):
|
||
hits = []
|
||
for n, d in docs.items():
|
||
t = visible(d) if use_visible else d
|
||
for m in re.finditer(pattern, t):
|
||
ctx = t[max(0, m.start() - 40):m.end() + 40].replace('\n', ' ')
|
||
hits.append('%s: ...%s...' % (n.split('/')[-1], ctx))
|
||
return hits
|
||
|
||
|
||
results = []
|
||
|
||
|
||
def check(cls, what, hits):
|
||
results.append((cls, what, hits))
|
||
|
||
|
||
# ---- residue: LaTeX syntax in what the reader sees
|
||
check('residue', 'backslash', scan(r'\\', CH))
|
||
check('residue', 'brace', scan(r'[{}]', CH))
|
||
check('residue', 'bracketed length', scan(
|
||
r'\[\s*-?\d*\.?\d+\s*(em|ex|pt|pc|in|mm|cm|bp|sp)\s*\](\{[^}]*\})?', CH))
|
||
|
||
# ---- escaping: generated markup turned into text
|
||
check('escaping', 'escaped tag', scan(r'</?[a-zA-Z]', X, use_visible=False))
|
||
check('escaping', 'unrestored placeholder', scan(r'@@BLOCK\d+@@', X, use_visible=False))
|
||
|
||
# ---- empty: a block with nothing, or only page markers, in it
|
||
MARK = r'<span epub:type="pagebreak"[^>]*>[^<]*</span>'
|
||
check('empty', 'empty or marker-only block', scan(
|
||
r'<(p|div|blockquote|section|li|figcaption)(\s[^>]*)?>(\s|%s)*</\1>' % MARK,
|
||
CH, use_visible=False))
|
||
|
||
# ---- links
|
||
ids = {n: set(re.findall(r'\bid="([^"]+)"', d)) for n, d in X.items()}
|
||
bad = []
|
||
for n, d in X.items():
|
||
for h in re.findall(r'href="([^"]+)"', d):
|
||
if h.startswith(('http:', 'https:', 'mailto:')) or h.endswith('.css'):
|
||
continue
|
||
p, _, f = h.partition('#')
|
||
t = posixpath.normpath(posixpath.join(posixpath.dirname(n), p)) if p else n
|
||
if t not in X or (f and f not in ids[t]):
|
||
bad.append('%s -> %s' % (n.split('/')[-1], h))
|
||
check('links', 'fragment resolving nowhere', bad)
|
||
|
||
# ---- images
|
||
names = set(z.namelist())
|
||
check('images', 'img without width+height',
|
||
['%s: %s' % (n.split('/')[-1], t[:90]) for n, d in X.items()
|
||
for t in re.findall(r'<img [^>]*>', d) if not ('width=' in t and 'height=' in t)])
|
||
check('images', 'img file missing',
|
||
['%s: %s' % (n.split('/')[-1], s) for n, d in X.items()
|
||
for s in re.findall(r'src="([^"]+)"', d)
|
||
if posixpath.normpath(posixpath.join(posixpath.dirname(n), s)) not in names])
|
||
|
||
# ---- xml
|
||
mal = []
|
||
for n in z.namelist():
|
||
if n.endswith(('.xhtml', '.opf', '.xml', '.ncx')):
|
||
try:
|
||
md.parseString(z.read(n))
|
||
except Exception as e: # noqa: BLE001
|
||
mal.append('%s: %s' % (n, str(e)[:70]))
|
||
check('xml', 'not well-formed', mal)
|
||
|
||
# ---- content: every source token must reach the EPUB (the project's own gate)
|
||
r = subprocess.run([sys.executable, 'tools/reprocheck.py', EPUB],
|
||
capture_output=True, text=True)
|
||
first = r.stdout.splitlines()[0] if r.stdout else r.stderr.strip()
|
||
m = re.search(r'(\d+) tokens short', first)
|
||
check('content', 'tokens lost (reprocheck)',
|
||
[] if m and m.group(1) == '0' else [first] + r.stdout.splitlines()[1:6])
|
||
|
||
# ---- typography: the EPUB carries every transform polish.py makes
|
||
import polish # noqa: E402
|
||
# Expected: everything the PDF sets, the way build.sh sets it -- the pages
|
||
# through polish.py, the colophon and errata raw. (The colophon's ProQuest
|
||
# address carries an en dash of its own.)
|
||
pol = re.sub(r'(?m)(?<!\\)%.*$', '',
|
||
''.join(polish.polish(open(f).read())
|
||
for f in sorted(glob.glob('src/pages/p*.tex')))
|
||
+ open('src/colophon.tex').read() + open('src/errata.tex').read())
|
||
body = ''.join(CH.values())
|
||
text = ''.join(visible(d) for d in CH.values())
|
||
want = {
|
||
'range en dashes': (len(re.findall(r'(?<!-)--(?!-)', pol)), text.count('–')),
|
||
'cross-reference links': (pol.count('\\xref{'), body.count('class="xref"')),
|
||
'unbreakable specimens': (pol.count('\\mbox{'), body.count('class="nowrap"')),
|
||
}
|
||
drift = ['%s: polish.py makes %d, EPUB has %d' % (k, a, b)
|
||
for k, (a, b) in want.items() if a != b]
|
||
ties = pol.count('~')
|
||
if text.count(' ') < ties:
|
||
drift.append('ties: polish.py makes %d, EPUB has %d no-break spaces'
|
||
% (ties, text.count(' ')))
|
||
check('typography', 'drift from the PDF pass', drift)
|
||
|
||
# ---- report
|
||
w = max(len(c) + len(t) for c, t, _ in results) + 3
|
||
failed = 0
|
||
for cls, what, hits in results:
|
||
ok = not hits
|
||
failed += not ok
|
||
print('[%s] %-*s %4d%s' % (' ok ' if ok else 'FAIL', w, cls + ': ' + what, len(hits),
|
||
'' if ok else ' e.g. ' + hits[0][:150]))
|
||
print('\n%d failing check(s)' % failed)
|
||
sys.exit(1 if failed else 0)
|