diff --git a/.gitignore b/.gitignore
index 2935934..1fc571d 100644
--- a/.gitignore
+++ b/.gitignore
@@ -23,3 +23,6 @@ src/main.tex
# Python bytecode, and per-session scratch work (backups, probes, OCR re-scan).
__pycache__/
.scratch/
+
+# The reflowable edition, like the PDF, is remade by `make epub`.
+ross-1988-retypeset.epub
diff --git a/Makefile b/Makefile
index bcca0c9..21b2429 100644
--- a/Makefile
+++ b/Makefile
@@ -11,7 +11,7 @@ SOURCES := build.sh tools/polish.py tools/gen_seal.py tools/gen_cover_mosaic.py
src/errata.tex $(wildcard src/pages/*.tex)
.DEFAULT_GOAL := pdf
-.PHONY: pdf build image prep plate cover check reprocheck verify clean distclean shell help
+.PHONY: pdf build image prep plate cover epub check reprocheck verify clean distclean shell help
# The PDF is a build product and is not in git. This remakes it from src/ and
# plates/, neither of which needs the ProQuest scan — only prep and plate do.
@@ -43,6 +43,10 @@ reprocheck: ## every token of src/pages must reach the built PDF
# so make would skip the rebuild and verify would pass against stale output.
verify: check build reprocheck ## the full gate: check, typeset unconditionally, then prove nothing was dropped
+epub: ## build the reflowable EPUB (needs the PDF built, for its cover), then audit it
+ $(RUN) python3 tools/make_epub.py
+ $(RUN) python3 tools/epub_audit.py
+
cover: ## rebuild the cover mosaic from plates/ and fonts/
$(RUN) python3 tools/gen_cover_mosaic.py --force
diff --git a/src/colophon.tex b/src/colophon.tex
index 5efed51..48e8588 100644
--- a/src/colophon.tex
+++ b/src/colophon.tex
@@ -5,19 +5,19 @@
\begingroup\small\setstretch{1.2}\setlength{\parindent}{0pt}\setlength{\parskip}{0.6em}
This is a re-typeset, searchable edition of Fiona G.\,E. Ross, \emph{The Evolution of the Printed Bengali Character from 1778 to 1978} (Ph.D. thesis, School of Oriental and African Studies, University of London, 1988), made in 2026 from the ProQuest scan of 431 leaves, ProQuest number 10731406. Every page was transcribed from the page image; the scan's OCR text layer was not used as a source.
-The text is reflowed and the original pagination kept: a number in the outer margin marks where each page of the 1988 thesis begins, and the running head gives the page range, so that the Contents, the List of Plates and the author's own cross-references still refer to the original numbering. The footnotes are set as endnotes, grouped by chapter and keeping their numbers.
+The text is reflowed and the original pagination kept: \byedition{a number in the outer margin marks where each page of the 1988 thesis begins, and the running head gives the page range}{a small number at the right-hand end of the line marks where each page of the 1988 thesis begins, and the same numbers make up the reader's page list}, so that the Contents, the List of Plates and the author's own cross-references still refer to the original numbering. The footnotes are set as endnotes, grouped by chapter and keeping their numbers.
The 178 plates are reproduced from the scan at its own 300 dpi, cropped clear of the page number, caption and scanner margins, with the captions re-set. The type specimens the author sets into the run of her own sentences are likewise cut from the scan rather than retyped, since it is their letterforms that the argument concerns.
Spelling and punctuation stand as printed, inconsistencies included. The author's own slips are corrected and listed in the Errata; slips inside quoted matter are left as printed, since they may belong to the source quoted. Corrections she made by hand in the scanned copy are adopted; the library ownership stamps are omitted.
-Set by XeLaTeX in XCharter, an extension of Matthew Carter's Charter, with Liberation Sans for the margin marks and running heads and Tiro Bangla for Bengali. The cover is set in EB Garamond, its opening line in XCharter. The mark below is generated from a hash of this edition's transcribed text and changes whenever that text does: \texttt{\sealseed}.
+\byedition{Set by XeLaTeX in XCharter, an extension of Matthew Carter's Charter, with Liberation Sans for the margin marks and running heads and Tiro Bangla for Bengali.}{The text face is the reader's to choose; Bengali is set in Tiro Bangla, which is embedded.} The cover is set in EB Garamond, its opening line in XCharter. The mark below is generated from a hash of this edition's transcribed text and changes whenever that text does: \texttt{\sealseed}.
\begin{center}
\begin{tikzpicture}[scale=0.62]\sealbody\end{tikzpicture}
\end{center}
-The ProQuest notice that precedes the title page in the scan is given overleaf.\par
+The ProQuest notice that precedes the title page in the scan is given \byedition{overleaf}{below}.\par
\endgroup
diff --git a/src/errata.tex b/src/errata.tex
index 9e7df84..9a15c22 100644
--- a/src/errata.tex
+++ b/src/errata.tex
@@ -1,4 +1,11 @@
% Errata: the original's own slips, corrected in this edition.
+% The introductory note lives here, not in \printerrata, so that the PDF and the
+% EPUB set the same words from one place.
+{\small Slips in the original typescript that have been corrected in this
+edition. The page number is the original's and links to the passage; the
+reading as typed is given first. Corrections the author made by hand in the
+scanned copy are adopted silently and are not listed here.\par}\medskip
+
% One \erratumline{original page}{as printed}{corrected} per \erratum{}{} in
% src/pages/; tools/check.py enforces the correspondence. Keep in page order.
\erratumline{10}{Navarnārī}{Navanārī}
diff --git a/src/preamble.tex b/src/preamble.tex
index d569524..ed7ced1 100644
--- a/src/preamble.tex
+++ b/src/preamble.tex
@@ -282,6 +282,11 @@
% src/errata.tex (enforced by tools/check.py), which \printerrata sets out.
\newwrite\erratafile
\newcommand{\erratum}[2]{#1\write\erratafile{#2 -> #1}}
+% \byedition{PDF wording}{EPUB wording}: a sentence of the shared front or back
+% matter that is only true of one edition -- the colophon's "running head", its
+% typefaces, "overleaf". LaTeX sets the first; tools/make_epub.py the second. One
+% source, so the sentences both editions share cannot drift apart.
+\newcommand{\byedition}[2]{#1}
\newcommand{\erratumline}[3]{\noindent\makebox[0.6in][l]{p.~\pg{#1}}%
\parbox[t]{\dimexpr\textwidth-0.6in\relax}{\raggedright
reads `#2'; corrected here to `#3'}\par\smallskip}
@@ -290,10 +295,6 @@
\fancyhead[R]{\small Errata}\fancyfoot[C]{\small\thepage}}
\newcommand{\printerrata}{\clearpage\pagestyle{errata}\section*{Errata}%
\pdfbookmark[0]{Errata}{sec:errata}%
- {\small Slips in the original typescript that have been corrected in this
- edition. The page number is the original's and links to the passage; the
- reading as typed is given first. Corrections the author made by hand in the
- scanned copy are adopted silently and are not listed here.\par}\medskip
\input{src/errata.tex}}
% ---- plates -----------------------------------------------------------------
diff --git a/tools/epub_audit.py b/tools/epub_audit.py
new file mode 100644
index 0000000..9e85913
--- /dev/null
+++ b/tools/epub_audit.py
@@ -0,0 +1,162 @@
+#!/usr/bin/env python3
+r"""Audit the EPUB by CLASS of fault, not by instance. `make epub` runs it and
+it exits non-zero on any failure.
+
+Each check targets a class of fault the rendering sweep of 2026-09-26 found at
+least one instance of. They are written against the class, so a new instance
+of an old fault fails here even if it looks nothing like the first one:
+
+ residue any LaTeX syntax reaching the reader -- a backslash, a brace, or
+ a bracketed length. (First instance: "\\[0.6em]" printed as text.)
+ escaping generated markup re-escaped into visible text. (First:
,
+
showing as text after a second conversion pass.)
+ empty a block element holding nothing, or only page markers -- a stray
+ blank line. (First: 31 empty
around page breaks.)
+ links a fragment link resolving to no id in the book. (First: 199
+ Contents and Plates links pointing into the wrong file.)
+ images an without width and height, or with a missing file.
+ xml any document that is not well-formed.
+ content any token of the transcription that does not reach the EPUB --
+ the class that holds silently dropped macros, dropped glyphs and
+ invisible page numbers alike. Runs the project's reprocheck.
+ typography the EPUB's typographic pass drifting from the PDF's: every range
+ en dash, tie, cross-reference link and unbreakable specimen that
+ tools/polish.py produces must reach the EPUB.
+
+Empty table cells are deliberately NOT a fault: the Scheme of Transliteration's
+last row is half-filled in the source itself.
+"""
+import glob
+import html
+import os
+import posixpath
+import re
+import subprocess
+import sys
+import xml.dom.minidom as md
+import zipfile
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+os.chdir(ROOT)
+sys.path.insert(0, os.path.join(ROOT, 'tools'))
+EPUB = sys.argv[1] if len(sys.argv) > 1 else 'ross-1988-retypeset.epub'
+
+z = zipfile.ZipFile(EPUB)
+X = {n: z.read(n).decode() for n in sorted(z.namelist()) if n.endswith('.xhtml')}
+# every document a reader reads -- not the navigation, not the cover image page
+CH = {n: d for n, d in X.items() if not re.search(r'/(nav|cover)\.xhtml$', n)}
+
+
+def visible(d):
+ d = re.sub(r'
'
+ % (pg, pg, self.convert(typed), self.convert(corrected))), i
+
+ def cmd_texttt(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_emph(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_textbf(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_textsuperscript(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_B(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_origpage(self, s, i):
+ a, i = balanced(s, i)
+ num = a.strip()
+ anchor = 'pg-%s' % num
+ self.pages.append((num, self.cur_file, anchor))
+ return pagebreak(anchor, num), i
+
+ def cmd_fn(self, s, i):
+ num, i = balanced(s, i)
+ body, i = balanced(s, i)
+ num = num.strip()
+ # Unique across the whole book, not just the chapter: every note now
+ # lands in one notes.xhtml, where two chapters' "note 1" sharing an id
+ # would send both references to the first.
+ nid = '%s-n%s-%d' % (os.path.splitext(self.cur_file or 'x')[0], num, len(self.notes))
+ self.notes.append((num, nid, self.convert(body)))
+ return ('%s' % (nid, nid, num)), i
+
+ def cmd_ig(self, s, i):
+ a, i = balanced(s, i)
+ name, w, h = epub_image(a.strip())
+ # same ratio as print: 1px = 1/300in, text ~12pt, so 1em = 1/6in
+ em = h / 50.0
+ return (''
+ % (name, w, h, em)), i
+
+ def cmd_fig(self, s, i):
+ a, i = balanced(s, i)
+ name, w, h = epub_image(a.strip())
+ return ('
' % (name, w, h)), i
+
+ def _plate(self, num, imgpath, caption, origpage=None):
+ name, w, h = epub_image(imgpath.strip())
+ pre = ''
+ if origpage is not None:
+ anchor = 'pg-%s' % origpage
+ self.pages.append((origpage, self.cur_file, anchor))
+ pre = pagebreak(anchor, origpage)
+ return ('%s'
+ ''
+ '%s. %s'
+ % (pre, num, name, num, w, h, num, self.convert(caption)))
+
+ def cmd_plateop(self, s, i):
+ pg, i = balanced(s, i)
+ num, i = balanced(s, i)
+ img, i = balanced(s, i)
+ cap, i = balanced(s, i)
+ return self._plate(num.strip(), img, cap, pg.strip()), i
+
+ def cmd_plate(self, s, i):
+ num, i = balanced(s, i)
+ img, i = balanced(s, i)
+ cap, i = balanced(s, i)
+ return self._plate(num.strip(), img, cap), i
+
+ def cmd_subhead(self, s, i):
+ a, i = balanced(s, i)
+ return '
' + self.convert(a) + '
', i
+
+ def cmd_subchap(self, s, i):
+ a, i = balanced(s, i)
+ return '
' + self.convert(a) + '
', i
+
+ def cmd_unsure(self, s, i):
+ a, i = balanced(s, i)
+ return '' + self.convert(a) + '', i
+
+ def cmd_qslip(self, s, i):
+ a, i = balanced(s, i)
+ return self.convert(a), i
+
+ def cmd_erratum(self, s, i):
+ good, i = balanced(s, i)
+ _bad, i = balanced(s, i)
+ return self.convert(good), i
+
+ def cmd_pg(self, s, i):
+ a, i = balanced(s, i)
+ num = a.strip()
+ return '%s' % (num, num), i
+
+ def cmd_pl(self, s, i):
+ num, i = balanced(s, i)
+ cap, i = balanced(s, i)
+ pg, i = balanced(s, i)
+ return ('
'
+ % (num.strip(), self.convert(cap), pg.strip(), pg.strip())), i
+
+ def cmd_plx(self, s, i):
+ """Like \\pl but its third argument is free markup (the List of Plates
+ header row), not a bare page number."""
+ num, i = balanced(s, i)
+ cap, i = balanced(s, i)
+ tail, i = balanced(s, i)
+ return ('
%s. %s %s
'
+ % (self.convert(num), self.convert(cap), self.convert(tail))), i
+
+ def cmd_tocl(self, s, i):
+ _ind, i = balanced(s, i)
+ label, i = balanced(s, i)
+ title, i = balanced(s, i)
+ pg, i = balanced(s, i)
+ return ('
'
+ % (self.convert(label), self.convert(title), pg.strip(), pg.strip())), i
+
+ def cmd_toclnp(self, s, i):
+ _ind, i = balanced(s, i)
+ label, i = balanced(s, i)
+ title, i = balanced(s, i)
+ return ('
%s %s
'
+ % (self.convert(label), self.convert(title))), i
+
+ def cmd_bibgroup(self, s, i):
+ a, i = balanced(s, i)
+ return '
' + self.convert(a) + '
', i
+
+ def cmd_bibhead(self, s, i):
+ a, i = balanced(s, i)
+ return '
' + self.convert(a) + '
', i
+
+ def cmd_XeTeXglyph(self, s, i):
+ """The transliteration table sets three isolated signs by raw
+ glyph index in Tiro Bangla, to keep the shaper from drawing a dotted
+ circle under them. A glyph index means nothing outside that font, so
+ each is mapped back to its character (read off the font's own cmap) and
+ set on a NO-BREAK SPACE -- Unicode's sanctioned way to show a combining
+ sign in isolation, which the Indic shapers accept as a base without
+ adding the dotted circle. Dropping them left three table cells empty.
+ """
+ m = re.match(r'\s*(\d+)', s[i:])
+ if not m:
+ return '', i
+ ch = GLYPH_CHARS.get(int(m.group(1)))
+ if ch is None:
+ raise LatexError('glyph %s has no character in the font' % m.group(1), s, i)
+ # only a mark needs the NBSP base; a spacing letter stands on its own
+ base = ' ' if unicodedata.category(ch).startswith('M') else ''
+ return '' + base + ch + '', i + m.end()
+
+
+# ---------------------------------------------------------------- signatures
+#
+# Every macro the (polished) transcription uses is declared here or has a
+# cmd_ handler. convert() raises on anything else, so a new macro in the source
+# is a build failure to be declared -- never text silently lost or leaked.
+
+# Layout only: consumed with exactly (optional, mandatory) arguments, emit nothing.
+LAYOUT = {
+ 'vspace': (0, 1), 'hspace': (0, 1), 'setlength': (0, 2), 'setstretch': (0, 1),
+ 'thispagestyle': (0, 1), 'pagestyle': (0, 1), 'pdfbookmark': (1, 2),
+ 'begingroup': (0, 0), 'endgroup': (0, 0), 'hfill': (0, 0), 'vfill': (0, 0),
+ 'medskip': (0, 0), 'smallskip': (0, 0), 'bigskip': (0, 0), 'noindent': (0, 0),
+ 'par': (0, 0), 'centering': (0, 0), 'clearpage': (0, 0), 'newpage': (0, 0),
+ 'bnsignfont': (0, 0), 'relax': (0, 0), 'protect': (0, 0),
+}
+
+# Declarations: no arguments; they restyle the rest of their group. A tag of
+# None means "size or weight the reflowable text leaves to the reader".
+DECLARATIONS = {
+ 'bfseries': 'strong', 'itshape': 'em', 'slshape': 'em', 'em': 'em',
+ 'scshape': '.smallcaps',
+ 'mdseries': None, 'upshape': None, 'normalfont': None, 'rmfamily': None,
+ 'small': None, 'footnotesize': None, 'scriptsize': None, 'normalsize': None,
+ 'large': None, 'Large': None, 'LARGE': None,
+}
+
+# A \\[..] argument is consumed only if it really is a length.
+LENGTH = re.compile(r'-?\d*\.?\d+\s*(em|ex|pt|pc|in|mm|cm|bp|sp|mu)')
+
+
+# A run of Bengali script, spaces inside it included. The print edition sets
+# ANY Bengali in Tiro Bangla (ucharclasses in src/preamble.tex); this is the
+# EPUB's version of that rule, so a Bengali word is never left to whatever
+# fallback font a reader has.
+BENGALI_RUN = re.compile(
+ '[ঀ-](?:[ঀ-]|[ ](?=[ঀ-]))*')
+
+
+class LatexError(Exception):
+ def __init__(self, msg, s, i):
+ ctx = s[max(0, i - 50):i + 50].replace('\n', ' ')
+ super().__init__('%s\n near: ...%s...' % (msg, ctx))
+
+
+CHAPTER_MACROS = ('chapstart', 'chaphead', 'chapnum', 'partstart',
+ 'sectionstart', 'matterstart')
+
+
+def plain(tex):
+ """A heading as plain text, for and the navigation: markup
+ stripped, line breaks become a space."""
+ c = Converter()
+ h = c.convert(tex).replace(' ', ' ')
+ return html.unescape(re.sub(r'<[^>]+>', '', h)).strip()
+
+
+def split_chapters(files):
+ """Group page files into chapters at the heading macros."""
+ chapters = []
+ cur = {'title': 'Front matter', 'short': 'Front matter', 'kind': 'front',
+ 'body': [], 'files': []}
+ for path in files:
+ # The polished text, not the raw transcription: tools/polish.py is the
+ # edition's typographic pass (range en dashes, ties, cross-reference
+ # links, unbreakable specimens), and the EPUB takes it from the same
+ # function the PDF build does, so a new rule reaches both editions.
+ raw = polish.polish(open(path).read())
+ pos = 0
+ while True:
+ m = re.search(r'\\(%s)' % '|'.join(CHAPTER_MACROS), raw[pos:])
+ if not m:
+ cur['body'].append(raw[pos:])
+ cur['files'].append(path)
+ break
+ start = pos + m.start()
+ cur['body'].append(raw[pos:start])
+ cur['files'].append(path)
+ j = start + m.end() - m.start()
+ cmd = m.group(1)
+ short = None
+ label = None
+ if cmd in ('chapstart', 'chaphead'):
+ short, j = opt_arg(raw, j)
+ title, j = balanced(raw, j)
+ elif cmd == 'chapnum':
+ a, j = balanced(raw, j)
+ b, j = balanced(raw, j)
+ # "Chapter 1" over "Charles Wilkins", as the print edition sets
+ # them -- joined by a line break, never by punctuation the
+ # original does not have (an earlier version invented ": ")
+ title = a + r'\\' + b
+ short = b
+ label = a
+ elif cmd == 'matterstart':
+ title, j = balanced(raw, j)
+ short = title
+ else: # partstart / sectionstart
+ a, j = balanced(raw, j)
+ b, j = balanced(raw, j)
+ title = a + r'\\' + b # was a + ' — ' + b: an invented dash
+ short = b
+ label = a
+ chapters.append(cur)
+ cur = {'title': title, 'short': short or title, 'kind': cmd,
+ 'label': label, 'body': [], 'files': [path]}
+ pos = j
+ chapters.append(cur)
+ return [c for c in chapters if ''.join(c['body']).strip()]
+
+
+ENV_RE = re.compile(r'\\begin\{(\w+)\}(.*?)\\end\{\1\}', re.S)
+
+
+def extract_envs(conv, text, blocks):
+ """Convert environments to HTML and stash them behind placeholders.
+
+ The generated markup must never go back through convert(), which escapes
+ < and >. An earlier version ran handle_envs() and then paragraphs() over
+ its output, so every
,
and
it had just produced was
+ re-escaped and surfaced as visible text in the reader. Placeholders keep
+ the two passes apart: the paragraph pass only ever sees LaTeX.
+ """
+ def repl(m):
+ env, inner = m.group(1), m.group(2)
+ if env == 'tikzpicture':
+ if '\\sealbody' not in inner:
+ raise LatexError('a tikzpicture other than the version mark', inner, 0)
+ key = '@@BLOCK%d@@' % len(blocks)
+ blocks[key] = ('
'
+ % (escape(SEAL['seed']), SEAL['w'], SEAL['h']))
+ return '\n\n' + key + '\n\n'
+ if env == 'tabular':
+ # the column spec contains nested braces (l@{}c@{}...), so a
+ # non-greedy {...} strip leaves "l@c@l@c@" glued to the first cell
+ k = 0
+ while k < len(inner) and inner[k] in ' \n\t':
+ k += 1
+ if k < len(inner) and inner[k] == '{':
+ _spec, k = balanced(inner, k)
+ inner = inner[k:]
+ rows = []
+ for line in inner.split('\\\\'):
+ line = line.strip()
+ if not line:
+ continue
+ cells = [conv.convert(c.strip()) for c in line.split('&')]
+ rows.append('
' + ''.join('
%s
' % c for c in cells) + '
')
+ body = '
' + ''.join(rows) + '
'
+ else:
+ inner = extract_envs(conv, inner, blocks)
+ if env == 'addmargin':
+ inner = re.sub(r'^\s*(\[[^\]]*\])?\s*(\{[^}]*\})?', '', inner)
+ body = paragraphs(conv, inner)
+ if env == 'extract':
+ body = '
' + body + '
'
+ elif env == 'biblist':
+ body = '
' + body + '
'
+ elif env == 'center':
+ body = '
' + body + '
'
+ elif env == 'addmargin':
+ body = '
' + body + '
'
+ key = '@@BLOCK%d@@' % len(blocks)
+ blocks[key] = body
+ return '\n\n' + key + '\n\n'
+
+ prev = None
+ while prev != text:
+ prev = text
+ text = ENV_RE.sub(repl, text)
+ return text
+
+
+def restore(text, blocks):
+ prev = None
+ while prev != text:
+ prev = text
+ for k, v in blocks.items():
+ text = text.replace(k, v)
+ return text
+
+
+BLOCK_START = (']*>[^<]*\s*)+')
+BLOCK_AFTER_MARKERS = re.compile(
+ r'(]*>[^<]*\s*)+(%s)' % '|'.join(
+ re.escape(b) for b in (' it renders as a stray blank line (31 of
+ # them), so hold it and set it at the head of whatever follows.
+ if PAGEBREAK_ONLY.fullmatch(h):
+ pending.append(h)
+ continue
+ if pending:
+ h = ''.join(pending) + h
+ pending = []
+ if h.startswith(BLOCK_START) or h.startswith('' + h + '')
+ out.extend(pending) # a trailing marker: keep, unwrapped
+ return '\n'.join(out)
+
+
+XHTML = """
+
+
+%s
+
+
+%s
+
+"""
+
+CSS = """/* The print edition's typography, carried wherever it survives reflow.
+ Source of each rule: src/preamble.tex. Disposition, rule by rule:
+
+ CARRIED
+ first-line indent 1.6em, 0.15em paragraph space -> as is
+ no indent after a heading or display -> as is
+ ragged right, no hyphenation (typescript breaks
+ only at spaces and typed hyphens) -> text-align:left, hyphens:none
+ widows and orphans forbidden -> widows/orphans 2
+ hanging punctuation -> hanging-punctuation (Apple
+ Books honours it; others ignore)
+ extract: indented both sides, set solid, no indent -> margins in em, tighter leading
+ bibliography: hanging indent 1.6em, set solid -> as is
+ cross-reference links: a hairline rule in linkink
+ (95,125,165), the text itself uncoloured -> underline, 1px, that colour
+ Contents / Plates whole-line links: unmarked -> as is
+ endnotes and plate captions a step smaller -> as is
+ specimens at native size (1px = 1/300in) -> height in em, per image
+ original page numbers in the outer margin, in sans -> floated to the right edge
+ of the line the page starts on
+ range en dashes, ties, specimen unbreakability -> from tools/polish.py itself
+
+ DROPPED, because reflow makes them meaningless or harmful
+ 13pt XCharter on a 5.95in measure the reader sets face, size and margins
+ 22.9pt absolute leading an absolute leading fights the reader's
+ own size; a relative 1.5 is kept instead
+ running heads there is no fixed page; readers show
+ their own, and the page-list carries
+ the original pagination for navigation
+ microtype expansion, raggedbottom no equivalent in CSS
+*/
+html { font-family: Georgia, "Times New Roman", serif; }
+body { margin: 0 5%; line-height: 1.5; text-align: left;
+ hyphens: none; -webkit-hyphens: none; }
+p { margin: 0 0 0.15em; text-indent: 1.6em; widows: 2; orphans: 2;
+ hanging-punctuation: first last; }
+h1 + p, h2 + p, h3 + p, blockquote + p, figure + p, table + p, div + p,
+.subhead + p, section > p:first-child { text-indent: 0; }
+h1, h2, h3 { font-weight: bold; line-height: 1.25; margin: 1.2em 0 0.6em;
+ page-break-after: avoid; }
+h1 { font-size: 1.5em; text-align: center; }
+h2 { font-size: 1.25em; }
+h3 { font-size: 1.1em; }
+.subhead { font-style: italic; margin-top: 1em; text-indent: 0; }
+.center { text-align: center; }
+.center p { text-indent: 0; }
+.inset { margin: 1em 1.5em; font-size: 0.9em; }
+.inset p { text-indent: 0; }
+
+/* extract: indented both sides (print 0.5in / 0.3in), set solid, unindented */
+blockquote { margin: 0.8em 1.5em 0.8em 2.5em; line-height: 1.35; }
+blockquote p { text-indent: 0; margin-bottom: 0.4em; }
+
+/* bibliography: solid, continuation lines hang 1.6em */
+.biblist p { text-indent: -1.6em; padding-left: 1.6em; margin: 0; }
+h3.bibgroup { margin-top: 1em; }
+h2.bibhead { text-align: center; }
+
+/* links: a hairline rule in the print edition's linkink, text uncoloured */
+a { color: inherit; }
+a.xref { text-decoration: underline; text-decoration-thickness: 1px;
+ text-decoration-color: rgb(95,125,165); text-underline-offset: 0.18em; }
+a.pgref, a.noteref { text-decoration: none; }
+
+/* Images: width and height attributes on every give the reader the real
+ aspect ratio; these rules let it scale without distorting -- the common EPUB
+ fault is a reader stretching an image to the frame. */
+img { max-width: 100%; height: auto; }
+figure.plate { margin: 1.2em 0; text-align: center; page-break-inside: avoid; }
+figure.plate img { max-width: 100%; max-height: 88vh; width: auto; height: auto; }
+figcaption { font-size: 0.85em; text-align: left; margin-top: 0.4em; }
+.figure { text-align: center; margin: 1em 0; }
+img.spec { vertical-align: -0.28em; width: auto; } /* height set per image */
+.nowrap { white-space: nowrap; }
+/* original page number: small sans at the right edge of the line the page
+ begins on, as the print edition sets it in the outer margin */
+.pgnum { float: right; clear: right; margin: 0.25em 0 0 0.6em;
+ font-family: "Liberation Sans", Arial, sans-serif; font-size: 0.65em;
+ font-weight: normal; font-style: normal; line-height: 1;
+ text-indent: 0; color: #8a8a8a; } /* \\mbox: never broken */
+
+/* Note numbers and other superscripts must not open up the line they sit on
+ -- the print edition's leading is even. A plain raises its baseline
+ and so enlarges the line box; lifting it by relative position instead
+ leaves the line box alone. */
+sup { font-size: 0.7em; line-height: 0; vertical-align: baseline;
+ position: relative; top: -0.5em; }
+
+/* Bengali in Tiro Bangla, embedded -- the print edition sets all Bengali in it */
+@font-face { font-family: "Tiro Bangla"; font-style: normal; font-weight: normal;
+ src: url(fonts/TiroBangla-Regular.ttf); }
+@font-face { font-family: "Tiro Bangla"; font-style: italic; font-weight: normal;
+ src: url(fonts/TiroBangla-Italic.ttf); }
+.bn, :lang(bn) { font-family: "Tiro Bangla", serif; }
+.bn { font-size: 1.15em; }
+.bn .bn { font-size: 1em; } /* nested runs must not compound */
+
+/* About this edition: set solid and unindented, a step smaller, as in print */
+.colophon p { text-indent: 0; margin: 0 0 0.6em; font-size: 0.92em; }
+.colophon .inset p, .colophon .inset { text-align: center; }
+.mark { text-align: center; margin: 1em 0; }
+/* the print mark stands about 2.5 text-heights tall; em keeps that proportion
+ at whatever size the reader sets, where its intrinsic 37px would not */
+.mark img { height: 2.6em; width: auto; max-width: 100%; }
+img.inline-graphic { max-height: 4em; width: auto; }
+
+/* Errata: the original's page in a column of its own, the entry hanging */
+.errata p { text-indent: 0; }
+p.erratum { padding-left: 4.5em; text-indent: -4.5em; margin: 0 0 0.35em; }
+p.erratum a.pgref { display: inline-block; min-width: 4.5em; text-indent: 0; }
+
+/* Notes: one section at the back, grouped by chapter, a step smaller */
+.notegroup h2 { font-size: 1.05em; margin: 1.4em 0 0.5em; }
+div.note p { text-indent: 0; margin: 0 0 0.35em; font-size: 0.9em; }
+a.noteback { text-decoration: none; }
+.smallcaps { font-variant: small-caps; }
+table.translit { border-collapse: collapse; margin: 1em 0; }
+table.translit td { padding: 0.15em 0.6em 0.15em 0; vertical-align: baseline; }
+.plentry, .tocl { margin: 0 0 0.3em; padding-left: 2em; text-indent: -2em; }
+.plnum { display: inline-block; min-width: 2.2em; }
+.unsure { border-bottom: 1px dotted #999; }
+.notes { margin-top: 2em; border-top: 1px solid #ccc; padding-top: 1em;
+ font-size: 0.9em; }
+.notes h2 { font-size: 1.1em; }
+.notes p { text-indent: 0; }
+aside.note { margin: 0 0 0.5em; }
+"""
+
+
+def finish_documents(names):
+ r"""Whole-book passes over the written chapters.
+
+ 1. Fragment links. Anything the converter links by id -- an original page
+ (\pg, \xref, the Contents and List of Plates), a plate, a note -- is
+ emitted as a same-document "#id", because at conversion time it cannot
+ know which chapter file the target lands in. Resolve every one against
+ an index of all ids in the book; one that resolves nowhere fails the
+ build. (199 dead Contents and Plates links came from resolving only the
+ targets that happened to share a file.)
+
+ 2. Empty blocks. A block holding nothing -- or only page-break markers --
+ renders as a stray blank line. Unwrap the markers (they stay, bare, in
+ the flow) and drop the empty block. Applies to every block element,
+ not only the
where it was first noticed.
+ """
+ docs = {n: open(os.path.join(OEBPS, n)).read() for n in names}
+ owner = {}
+ for n, d in docs.items():
+ for i in re.findall(r'\bid="([^"]+)"', d):
+ owner.setdefault(i, n)
+ marker = r']*>[^<]*'
+ empty = re.compile(r'<(p|div|blockquote|section|li|figcaption)(\s[^>]*)?>'
+ r'((?:\s|%s)*)\1>' % marker)
+ for n, d in docs.items():
+ def fix(m, n=n):
+ frag = m.group(1)
+ if re.search(r'\bid="%s"' % re.escape(frag), docs[n]):
+ return m.group(0)
+ if frag not in owner:
+ raise SystemExit('make_epub: link to #%s in %s resolves nowhere' % (frag, n))
+ return 'href="%s#%s"' % (owner[frag], frag)
+ d = re.sub(r'href="#([^"]+)"', fix, d)
+ prev = None
+ while prev != d:
+ prev = d
+ d = empty.sub(lambda m: m.group(3).strip(), d)
+ open(os.path.join(OEBPS, n), 'w').write(d)
+
+
+ROMAN = ['I', 'II', 'III', 'IV', 'V', 'VI', 'VII', 'VIII', 'IX', 'X']
+
+
+def nav_items(heads):
+ """(level, label, file) for each chapter, as the PDF outline gives them.
+
+ The print edition's outline nests Parts > Sections > Chapters and labels
+ them by position -- I, I.A, I.A.1; II.8 where a Part has no Sections --
+ keeping the author's chapter numbers (src/preamble.tex, "navigation
+ numbering"). The same labels head the PDF's note groups ("Notes to I.A.1
+ Charles Wilkins"), so both the outline and the notes take them from here.
+ """
+ items = []
+ part = sec = 0
+ for fn, ch in heads:
+ k = ch['kind']
+ name = plain(ch['short'])
+ if k == 'front':
+ continue
+ if k == 'partstart':
+ part += 1; sec = 0
+ items.append((1, '%s %s' % (ROMAN[part - 1], name), fn))
+ elif k == 'sectionstart':
+ sec += 1
+ items.append((2, '%s.%s %s' % (ROMAN[part - 1], chr(64 + sec), name), fn))
+ elif k == 'chapnum':
+ num = re.search(r'\d+', ch['label'] or '').group(0)
+ pos = ROMAN[part - 1] + ('.' + chr(64 + sec) if sec else '')
+ items.append((3 if sec else 2, '%s.%s %s' % (pos, num, name), fn))
+ else: # unnumbered: top level
+ items.append((1, name, fn))
+ return items
+
+
+def outline(items):
+ return _nest(items, 0, 1)[0]
+
+
+def _nest(items, i, level):
+ """items[i:] as
elements at `level`, deeper items nested inside the
+
that precedes them. Returns (html, next index)."""
+ out = []
+ while i < len(items) and items[i][0] >= level:
+ _lvl, text, fn = items[i]
+ out.append('
%s' % (fn, escape(text)))
+ i += 1
+ if i < len(items) and items[i][0] > level:
+ inner, i = _nest(items, i, level + 1)
+ out.append('' + inner + '')
+ out.append('
')
+ return ''.join(out), i
+
+
+def build_mark():
+ r"""Render the version mark to SVG, from tools/gen_seal.py itself.
+
+ The PDF draws the mark with TikZ from a hash of the sources; the EPUB takes
+ the same seed and the same drawing code and renders that TikZ to SVG
+ (XeLaTeX, then pdftocairo). Vector, so it holds at any size, and the same
+ mark as the PDF's whenever both are built from the same sources.
+ """
+ import gen_seal
+ seed = gen_seal.content_hash()
+ body = gen_seal.build(seed)
+ tmp = os.path.join(BUILD, 'mark')
+ os.makedirs(tmp, exist_ok=True)
+ with open(os.path.join(tmp, 'mark.tex'), 'w') as f:
+ f.write('\\documentclass[tikz,border=2pt]{standalone}\n\\usepackage{xcolor}\n'
+ '\\definecolor{ink}{RGB}{26,26,28}\\definecolor{paper}{RGB}{255,255,255}\n'
+ '\\begin{document}\\begin{tikzpicture}[scale=0.62]\n%s\n'
+ '\\end{tikzpicture}\\end{document}\n' % body)
+ subprocess.run(['xelatex', '-interaction=nonstopmode', 'mark.tex'], cwd=tmp,
+ check=True, stdout=subprocess.DEVNULL)
+ svg = os.path.join(IMG, 'mark.svg')
+ subprocess.run(['pdftocairo', '-svg', os.path.join(tmp, 'mark.pdf'), svg], check=True)
+ m = re.search(r'