tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.
\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
1245 lines
52 KiB
Python
1245 lines
52 KiB
Python
#!/usr/bin/env python3
|
||
r"""Build a reflowable EPUB 3 from the same src/pages transcription as the PDF.
|
||
|
||
docker compose run --rm -T tex python3 tools/make_epub.py
|
||
|
||
WHAT TRANSFERS AND WHAT DOES NOT
|
||
|
||
The PDF's typography is fixed-measure work -- a 22.9pt leading anchored to the
|
||
inline specimens, a 5.95in measure, hanging punctuation, ragged-right with
|
||
hyphenation off, plates scaled to the measure. None of that survives reflow and
|
||
none of it is attempted here. What the EPUB keeps is the edition's substance:
|
||
the text, the original pagination, the plates and inline specimens, the
|
||
endnotes, and the editorial apparatus.
|
||
|
||
PAGE NUMBERS
|
||
|
||
The 1988 pagination is the spine of this edition -- the Contents, the List of
|
||
Plates and the author's own cross-references all cite it. Every \origpage{N}
|
||
becomes an EPUB 3 pagebreak marker, and all of them are collected into the
|
||
page-list of the navigation document, which is what makes a reader's "go to
|
||
page" and page-number display work on a reflowable book. dc:source and
|
||
pageBreakSource name the print original the numbering belongs to.
|
||
|
||
IMAGES
|
||
|
||
Every <img> carries explicit width and height attributes taken from the file's
|
||
real pixel dimensions, so a reader that lays out before the image loads still
|
||
reserves the right box. The CSS sets width:auto/height:auto against those, which
|
||
is what stops the common EPUB failure of images stretched to the frame.
|
||
|
||
Inline type specimens keep their relative sizes: \ig sets them at 1px = 1/300in
|
||
in print, so here each gets a height in em derived from the same ratio -- a
|
||
vowel sign must not come out as tall as a letter.
|
||
"""
|
||
import glob
|
||
import html
|
||
import os
|
||
import re
|
||
import shutil
|
||
import subprocess
|
||
import sys
|
||
import unicodedata
|
||
import zipfile
|
||
from xml.sax.saxutils import escape
|
||
|
||
import numpy as np
|
||
from PIL import Image
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
# polish.py builds its plate/chapter maps from the relative path src/pages at
|
||
# import time, so it must be imported from the project root.
|
||
os.chdir(ROOT)
|
||
sys.path.insert(0, os.path.join(ROOT, 'tools'))
|
||
import polish # noqa: E402
|
||
BUILD = os.path.join(ROOT, 'work', 'epub')
|
||
OEBPS = os.path.join(BUILD, 'OEBPS')
|
||
IMG = os.path.join(OEBPS, 'images')
|
||
OUT = os.path.join(ROOT, 'ross-1988-retypeset.epub')
|
||
|
||
TITLE = 'The Evolution of the Printed Bengali Character from 1778 to 1978'
|
||
AUTHOR = 'Fiona Georgina Elisabeth Ross'
|
||
UID = 'urn:uuid:ross-1988-bengali-retypeset-2026'
|
||
|
||
|
||
# ---------------------------------------------------------------- helpers
|
||
|
||
def balanced(s, i):
|
||
"""Return (content, index after closing brace) for a { } group at s[i]=='{'."""
|
||
assert s[i] == '{'
|
||
depth = 0
|
||
j = i
|
||
while j < len(s):
|
||
if s[j] == '{':
|
||
depth += 1
|
||
elif s[j] == '}':
|
||
depth -= 1
|
||
if depth == 0:
|
||
return s[i + 1:j], j + 1
|
||
j += 1
|
||
return s[i + 1:], len(s)
|
||
|
||
|
||
def opt_arg(s, i):
|
||
"""Optional [..] argument."""
|
||
if i < len(s) and s[i] == '[':
|
||
j = s.index(']', i)
|
||
return s[i + 1:j], j + 1
|
||
return None, i
|
||
|
||
|
||
ESCAPED = {'&': '&', '%': '%', '$': '$', '#': '#', '_': '_',
|
||
' ': ' ', ',': '\u2009', '{': '{', '}': '}'}
|
||
|
||
# Letter-name macros that stand for text. Without these they fall through to
|
||
# the "unknown macro" path and are dropped silently -- \ldots alone accounts
|
||
# for 118 ellipses in this text.
|
||
TEXT_MACROS = {
|
||
'ldots': '\u2026', 'dots': '\u2026',
|
||
'textasciitilde': '~', 'textcopyright': '\u00a9',
|
||
'textemdash': '\u2014', 'textendash': '\u2013',
|
||
'textquoteleft': '\u2018', 'textquoteright': '\u2019',
|
||
'textquotedblleft': '\u201c', 'textquotedblright': '\u201d',
|
||
'pounds': '\u00a3', 'dag': '\u2020', 'ddag': '\u2021',
|
||
'LaTeX': 'LaTeX', 'TeX': 'TeX', 'XeLaTeX': 'XeLaTeX',
|
||
}
|
||
|
||
|
||
def _glyph_table():
|
||
r"""Glyph index -> character for Tiro Bangla, read off the font's own cmap.
|
||
|
||
The transcription sets some isolated signs by raw glyph index. Deriving the
|
||
table from the font, rather than typing in the three indices the table uses
|
||
today, means any \XeTeXglyph the source ever uses resolves -- or fails the
|
||
build if the font has no character for it.
|
||
"""
|
||
from fontTools.ttLib import TTFont
|
||
font = TTFont(os.path.join(ROOT, 'fonts', 'TiroBangla-Regular.ttf'))
|
||
order = font.getGlyphOrder()
|
||
first = {}
|
||
for cp, g in sorted(font.getBestCmap().items()):
|
||
first.setdefault(g, cp)
|
||
return {gid: chr(first[g]) for gid, g in enumerate(order) if g in first}
|
||
|
||
|
||
GLYPH_CHARS = _glyph_table()
|
||
|
||
IMG_CACHE = {}
|
||
|
||
|
||
def img_size(path):
|
||
if path not in IMG_CACHE:
|
||
with Image.open(os.path.join(ROOT, path)) as im:
|
||
IMG_CACHE[path] = im.size
|
||
return IMG_CACHE[path]
|
||
|
||
|
||
def epub_image(src_rel):
|
||
"""Copy/convert an image into the EPUB and return (name, w, h).
|
||
|
||
Bilevel plates go to 1-bit PNG and halftone plates to JPEG, the same split
|
||
the PDF uses -- an 8-bit greyscale PNG of a page of type is mostly storing
|
||
scanner grain.
|
||
"""
|
||
name = src_rel.replace('/', '_')
|
||
dst = os.path.join(IMG, name)
|
||
if os.path.exists(dst):
|
||
return name, *img_size(src_rel)
|
||
a = np.array(Image.open(os.path.join(ROOT, src_rel)).convert('L'))
|
||
h = np.bincount(a.ravel(), minlength=256)
|
||
mid = 1.0 - (h[:40].sum() + h[216:].sum()) / a.size
|
||
if src_rel.startswith('plates/inline/'):
|
||
Image.fromarray(a).save(dst, 'PNG', optimize=True)
|
||
elif mid > 0.25:
|
||
name = os.path.splitext(name)[0] + '.jpg'
|
||
dst = os.path.join(IMG, name)
|
||
Image.fromarray(a).save(dst, 'JPEG', quality=85, optimize=True)
|
||
else:
|
||
Image.fromarray(a > 128).save(dst, 'PNG', optimize=True, bits=1)
|
||
w, hh = a.shape[1], a.shape[0]
|
||
IMG_CACHE[src_rel] = (w, hh)
|
||
return name, w, hh
|
||
|
||
|
||
# ---------------------------------------------------------------- text
|
||
|
||
def textify(s):
|
||
"""LaTeX text conventions -> HTML entities and Unicode."""
|
||
s = s.replace('\\&', '&').replace('\\%', '%').replace('\\$', '$')
|
||
s = s.replace('\\#', '#').replace('\\_', '_')
|
||
s = s.replace('\\textasciitilde', '~')
|
||
s = s.replace('\\ldots\\ ', '… ').replace('\\ldots', '…')
|
||
s = s.replace('\\textcopyright{}', '©').replace('\\textcopyright', '©')
|
||
s = s.replace('\\,', '\u2009').replace('\\ ', ' ')
|
||
s = re.sub(r'``', '\u201c', s)
|
||
s = re.sub(r"''", '\u201d', s)
|
||
s = re.sub(r'`', '\u2018', s)
|
||
s = re.sub(r"(?<=[\w.,;:!?\u2019])'", '\u2019', s)
|
||
s = s.replace('---', '\u2014').replace('--', '\u2013')
|
||
s = s.replace('~', '\u00a0')
|
||
return s
|
||
|
||
|
||
def pagebreak(anchor, num):
|
||
"""An original-page marker, with the number as visible text.
|
||
|
||
The edition's whole reference apparatus -- the Contents, the List of
|
||
Plates, the author's own cross-references -- cites the 1988 pagination,
|
||
and the print edition sets each page's number in the outer margin so a
|
||
reader can see where a cited page begins. An empty marker carried the
|
||
numbers to the page-list, which serves navigation, but never to the page
|
||
itself; reprocheck counted all 128 of them missing. The number is floated
|
||
to the right edge of the line the page begins on -- reflow's equivalent of
|
||
the outer margin.
|
||
"""
|
||
return ('<span epub:type="pagebreak" role="doc-pagebreak" id="%s" '
|
||
'aria-label="%s" class="pgnum">%s</span>' % (anchor, num, num))
|
||
|
||
|
||
class Converter:
|
||
def __init__(self):
|
||
self.notes = [] # (n, html) for the current chapter
|
||
self.pages = [] # (page number, chapter file, anchor) in order
|
||
self.cur_file = None
|
||
|
||
def convert(self, s):
|
||
r"""LaTeX -> XHTML for one run of text.
|
||
|
||
Every control sequence is dispatched by a DECLARED signature and
|
||
consumes exactly the arguments that signature names -- no more, no
|
||
fewer. Anything undeclared stops the build (LatexError) instead of
|
||
being dropped. The earlier converter guessed: it dropped unknown macros
|
||
silently (118 ellipses went that way), and a generic "eat a following
|
||
[..]" rule would have deleted the editorial "[with]" that follows
|
||
\ldots in plate 6's caption. Signatures make both impossible.
|
||
"""
|
||
out = []
|
||
i = 0
|
||
n = len(s)
|
||
while i < n:
|
||
c = s[i]
|
||
if c == '\\':
|
||
m = re.match(r'\\([a-zA-Z]+)(\*?)', s[i:])
|
||
if not m: # control symbol
|
||
sym = s[i + 1:i + 2]
|
||
if sym == '\\': # \\ *? [length]?
|
||
out.append('<br/>')
|
||
k = i + 2
|
||
if s[k:k + 1] == '*':
|
||
k += 1
|
||
m2 = re.match(r'[ \t]*\[[^\]]*\]', s[k:])
|
||
if m2 and LENGTH.fullmatch(m2.group(0).strip()[1:-1].strip()):
|
||
k += m2.end()
|
||
i = k
|
||
continue
|
||
if sym in ESCAPED:
|
||
out.append(ESCAPED[sym])
|
||
i += 2
|
||
continue
|
||
raise LatexError('undeclared control symbol \\%s' % sym, s, i)
|
||
cmd = m.group(1)
|
||
j = i + m.end()
|
||
handler = getattr(self, 'cmd_' + cmd, None)
|
||
if handler:
|
||
txt, j = handler(s, j)
|
||
out.append(txt)
|
||
i = j
|
||
continue
|
||
if cmd in TEXT_MACROS:
|
||
out.append(TEXT_MACROS[cmd])
|
||
if s[j:j + 2] == '\\ ': # \ldots\ -- a real space
|
||
out.append(' ')
|
||
j += 2
|
||
i = j
|
||
continue
|
||
if cmd in DECLARATIONS:
|
||
# A declaration styles the rest of its group. convert() is
|
||
# called per group, so "the rest of the group" is the rest
|
||
# of s -- wherever in the group the switch falls, not only
|
||
# when it happens to come first.
|
||
rest = self.convert(s[j:])
|
||
tag = DECLARATIONS[cmd]
|
||
if tag:
|
||
rest = rest.lstrip()
|
||
rest = ('<span class="%s">%s</span>' % (tag[1:], rest)
|
||
if tag.startswith('.') else
|
||
'<%s>%s</%s>' % (tag, rest, tag))
|
||
out.append(rest)
|
||
i = n
|
||
continue
|
||
if cmd in LAYOUT:
|
||
nopt, nman = LAYOUT[cmd]
|
||
for _ in range(nopt):
|
||
_o, j = opt_arg(s, j)
|
||
for _ in range(nman):
|
||
_a, j = balanced(s, j)
|
||
i = j
|
||
continue
|
||
raise LatexError('undeclared macro \\%s' % cmd, s, i)
|
||
elif c == '{':
|
||
grp, j = balanced(s, i)
|
||
out.append(self.convert(grp))
|
||
i = j
|
||
elif c == '}':
|
||
i += 1
|
||
else:
|
||
k = i
|
||
while k < n and s[k] not in '\\{}':
|
||
k += 1
|
||
out.append(BENGALI_RUN.sub(
|
||
r'<span class="bn" lang="bn">\g<0></span>', escape(textify(s[i:k]))))
|
||
i = k
|
||
return ''.join(out)
|
||
|
||
# ---- macros
|
||
|
||
def cmd_newcommand(self, s, i):
|
||
"""Swallow a local macro definition. p0015/p0016 define \\B for the
|
||
transliteration table; without this the definition's own body is
|
||
rendered as if it were text."""
|
||
_name, i = balanced(s, i)
|
||
_n, i = opt_arg(s, i)
|
||
_body, i = balanced(s, i)
|
||
return '', i
|
||
|
||
cmd_renewcommand = cmd_newcommand
|
||
|
||
def cmd_xref(self, s, i):
|
||
r'''polish.py's cross-reference: \xref{page.N}{shown}. The target is an
|
||
original page, whose anchor is resolved across files after assembly.'''
|
||
target, i = balanced(s, i)
|
||
shown, i = balanced(s, i)
|
||
m = re.fullmatch(r'\s*page\.(\d+)\s*', target)
|
||
if not m:
|
||
raise LatexError('xref to unknown target %r' % target, s, i)
|
||
return '<a class="xref" href="#pg-%s">%s</a>' % (m.group(1), self.convert(shown)), i
|
||
|
||
cmd_hyperlink = cmd_xref
|
||
|
||
def cmd_mbox(self, s, i):
|
||
r'''\mbox keeps its content on one line. polish.py wraps every inline
|
||
specimen in one so it never parts from adjacent punctuation; on a
|
||
narrow reflowed screen that matters more, not less. Its content is
|
||
text -- the old drop-list would have discarded all 411 specimens.'''
|
||
a, i = balanced(s, i)
|
||
return '<span class="nowrap">' + self.convert(a) + '</span>', i
|
||
|
||
def cmd_byedition(self, s, i):
|
||
"""\byedition{PDF wording}{EPUB wording}: take the EPUB's."""
|
||
_pdf, i = balanced(s, i)
|
||
epub, i = balanced(s, i)
|
||
return self.convert(epub), i
|
||
|
||
def cmd_sealseed(self, s, i):
|
||
return escape(SEAL['seed']), i
|
||
|
||
def cmd_includegraphics(self, s, i):
|
||
_opt, i = opt_arg(s, i)
|
||
path, i = balanced(s, i)
|
||
path = path.strip()
|
||
name = os.path.basename(path)
|
||
dst = os.path.join(IMG, name)
|
||
if not os.path.exists(dst):
|
||
shutil.copy(os.path.join(ROOT, path), dst)
|
||
w, h = img_size(path)
|
||
return ('<img class="inline-graphic" src="images/%s" alt="" width="%d" '
|
||
'height="%d"/>' % (name, w, h)), i
|
||
|
||
def cmd_erratumline(self, s, i):
|
||
pg, i = balanced(s, i)
|
||
typed, i = balanced(s, i)
|
||
corrected, i = balanced(s, i)
|
||
pg = pg.strip()
|
||
return ('<p class="erratum"><a class="pgref" href="#pg-%s">p. %s</a> '
|
||
'reads ‘%s’; corrected here to ‘%s’</p>'
|
||
% (pg, pg, self.convert(typed), self.convert(corrected))), i
|
||
|
||
def cmd_texttt(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<code>' + self.convert(a) + '</code>', i
|
||
|
||
def cmd_emph(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<em>' + self.convert(a) + '</em>', i
|
||
|
||
def cmd_textbf(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<strong>' + self.convert(a) + '</strong>', i
|
||
|
||
def cmd_textsuperscript(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<sup>' + self.convert(a) + '</sup>', i
|
||
|
||
def cmd_B(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<span class="bn">' + self.convert(a) + '</span>', i
|
||
|
||
def cmd_origpage(self, s, i):
|
||
a, i = balanced(s, i)
|
||
num = a.strip()
|
||
anchor = 'pg-%s' % num
|
||
self.pages.append((num, self.cur_file, anchor))
|
||
return pagebreak(anchor, num), i
|
||
|
||
def cmd_fn(self, s, i):
|
||
num, i = balanced(s, i)
|
||
body, i = balanced(s, i)
|
||
num = num.strip()
|
||
# Unique across the whole book, not just the chapter: every note now
|
||
# lands in one notes.xhtml, where two chapters' "note 1" sharing an id
|
||
# would send both references to the first.
|
||
nid = '%s-n%s-%d' % (os.path.splitext(self.cur_file or 'x')[0], num, len(self.notes))
|
||
self.notes.append((num, nid, self.convert(body)))
|
||
return ('<a class="noteref" epub:type="noteref" role="doc-noteref" '
|
||
'href="#%s" id="%sr"><sup>%s</sup></a>' % (nid, nid, num)), i
|
||
|
||
def cmd_ig(self, s, i):
|
||
a, i = balanced(s, i)
|
||
name, w, h = epub_image(a.strip())
|
||
# same ratio as print: 1px = 1/300in, text ~12pt, so 1em = 1/6in
|
||
em = h / 50.0
|
||
return ('<img class="spec" src="images/%s" alt="type specimen" '
|
||
'width="%d" height="%d" style="height:%.2fem"/>'
|
||
% (name, w, h, em)), i
|
||
|
||
def cmd_fig(self, s, i):
|
||
a, i = balanced(s, i)
|
||
name, w, h = epub_image(a.strip())
|
||
return ('<div class="figure"><img src="images/%s" alt="" width="%d" '
|
||
'height="%d"/></div>' % (name, w, h)), i
|
||
|
||
def _plate(self, num, imgpath, caption, origpage=None):
|
||
name, w, h = epub_image(imgpath.strip())
|
||
pre = ''
|
||
if origpage is not None:
|
||
anchor = 'pg-%s' % origpage
|
||
self.pages.append((origpage, self.cur_file, anchor))
|
||
pre = pagebreak(anchor, origpage)
|
||
return ('%s<figure class="plate" id="plate-%s">'
|
||
'<img src="images/%s" alt="Plate %s" width="%d" height="%d"/>'
|
||
'<figcaption>%s. %s</figcaption></figure>'
|
||
% (pre, num, name, num, w, h, num, self.convert(caption)))
|
||
|
||
def cmd_plateop(self, s, i):
|
||
pg, i = balanced(s, i)
|
||
num, i = balanced(s, i)
|
||
img, i = balanced(s, i)
|
||
cap, i = balanced(s, i)
|
||
return self._plate(num.strip(), img, cap, pg.strip()), i
|
||
|
||
def cmd_plate(self, s, i):
|
||
num, i = balanced(s, i)
|
||
img, i = balanced(s, i)
|
||
cap, i = balanced(s, i)
|
||
return self._plate(num.strip(), img, cap), i
|
||
|
||
def cmd_subhead(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<p class="subhead"><em>' + self.convert(a) + '</em></p>', i
|
||
|
||
def cmd_subchap(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<h3>' + self.convert(a) + '</h3>', i
|
||
|
||
def cmd_unsure(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<span class="unsure">' + self.convert(a) + '</span>', i
|
||
|
||
def cmd_qslip(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return self.convert(a), i
|
||
|
||
def cmd_erratum(self, s, i):
|
||
good, i = balanced(s, i)
|
||
_bad, i = balanced(s, i)
|
||
return self.convert(good), i
|
||
|
||
def cmd_pg(self, s, i):
|
||
a, i = balanced(s, i)
|
||
num = a.strip()
|
||
return '<a class="pgref" href="#pg-%s">%s</a>' % (num, num), i
|
||
|
||
def cmd_pl(self, s, i):
|
||
num, i = balanced(s, i)
|
||
cap, i = balanced(s, i)
|
||
pg, i = balanced(s, i)
|
||
return ('<p class="plentry"><span class="plnum">%s.</span> %s '
|
||
'<a class="pgref" href="#pg-%s">%s</a></p>'
|
||
% (num.strip(), self.convert(cap), pg.strip(), pg.strip())), i
|
||
|
||
def cmd_plx(self, s, i):
|
||
"""Like \\pl but its third argument is free markup (the List of Plates
|
||
header row), not a bare page number."""
|
||
num, i = balanced(s, i)
|
||
cap, i = balanced(s, i)
|
||
tail, i = balanced(s, i)
|
||
return ('<p class="plentry"><span class="plnum">%s.</span> %s %s</p>'
|
||
% (self.convert(num), self.convert(cap), self.convert(tail))), i
|
||
|
||
def cmd_tocl(self, s, i):
|
||
_ind, i = balanced(s, i)
|
||
label, i = balanced(s, i)
|
||
title, i = balanced(s, i)
|
||
pg, i = balanced(s, i)
|
||
return ('<p class="tocl">%s %s <a class="pgref" href="#pg-%s">%s</a></p>'
|
||
% (self.convert(label), self.convert(title), pg.strip(), pg.strip())), i
|
||
|
||
def cmd_toclnp(self, s, i):
|
||
_ind, i = balanced(s, i)
|
||
label, i = balanced(s, i)
|
||
title, i = balanced(s, i)
|
||
return ('<p class="tocl">%s %s</p>'
|
||
% (self.convert(label), self.convert(title))), i
|
||
|
||
def cmd_bibgroup(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<h3 class="bibgroup">' + self.convert(a) + '</h3>', i
|
||
|
||
def cmd_bibhead(self, s, i):
|
||
a, i = balanced(s, i)
|
||
return '<h2 class="bibhead">' + self.convert(a) + '</h2>', i
|
||
|
||
def cmd_XeTeXglyph(self, s, i):
|
||
"""The transliteration table sets three isolated signs by raw
|
||
glyph index in Tiro Bangla, to keep the shaper from drawing a dotted
|
||
circle under them. A glyph index means nothing outside that font, so
|
||
each is mapped back to its character (read off the font's own cmap) and
|
||
set on a NO-BREAK SPACE -- Unicode's sanctioned way to show a combining
|
||
sign in isolation, which the Indic shapers accept as a base without
|
||
adding the dotted circle. Dropping them left three table cells empty.
|
||
"""
|
||
m = re.match(r'\s*(\d+)', s[i:])
|
||
if not m:
|
||
return '', i
|
||
ch = GLYPH_CHARS.get(int(m.group(1)))
|
||
if ch is None:
|
||
raise LatexError('glyph %s has no character in the font' % m.group(1), s, i)
|
||
# only a mark needs the NBSP base; a spacing letter stands on its own
|
||
base = ' ' if unicodedata.category(ch).startswith('M') else ''
|
||
return '<span class="bn">' + base + ch + '</span>', i + m.end()
|
||
|
||
|
||
# ---------------------------------------------------------------- signatures
|
||
#
|
||
# Every macro the (polished) transcription uses is declared here or has a
|
||
# cmd_ handler. convert() raises on anything else, so a new macro in the source
|
||
# is a build failure to be declared -- never text silently lost or leaked.
|
||
|
||
# Layout only: consumed with exactly (optional, mandatory) arguments, emit nothing.
|
||
LAYOUT = {
|
||
'vspace': (0, 1), 'hspace': (0, 1), 'setlength': (0, 2), 'setstretch': (0, 1),
|
||
'thispagestyle': (0, 1), 'pagestyle': (0, 1), 'pdfbookmark': (1, 2),
|
||
'begingroup': (0, 0), 'endgroup': (0, 0), 'hfill': (0, 0), 'vfill': (0, 0),
|
||
'medskip': (0, 0), 'smallskip': (0, 0), 'bigskip': (0, 0), 'noindent': (0, 0),
|
||
'par': (0, 0), 'centering': (0, 0), 'clearpage': (0, 0), 'newpage': (0, 0),
|
||
'bnsignfont': (0, 0), 'relax': (0, 0), 'protect': (0, 0),
|
||
}
|
||
|
||
# Declarations: no arguments; they restyle the rest of their group. A tag of
|
||
# None means "size or weight the reflowable text leaves to the reader".
|
||
DECLARATIONS = {
|
||
'bfseries': 'strong', 'itshape': 'em', 'slshape': 'em', 'em': 'em',
|
||
'scshape': '.smallcaps',
|
||
'mdseries': None, 'upshape': None, 'normalfont': None, 'rmfamily': None,
|
||
'small': None, 'footnotesize': None, 'scriptsize': None, 'normalsize': None,
|
||
'large': None, 'Large': None, 'LARGE': None,
|
||
}
|
||
|
||
# A \\[..] argument is consumed only if it really is a length.
|
||
LENGTH = re.compile(r'-?\d*\.?\d+\s*(em|ex|pt|pc|in|mm|cm|bp|sp|mu)')
|
||
|
||
|
||
# A run of Bengali script, spaces inside it included. The print edition sets
|
||
# ANY Bengali in Tiro Bangla (ucharclasses in src/preamble.tex); this is the
|
||
# EPUB's version of that rule, so a Bengali word is never left to whatever
|
||
# fallback font a reader has.
|
||
BENGALI_RUN = re.compile(
|
||
'[ঀ-](?:[ঀ-]|[ ](?=[ঀ-]))*')
|
||
|
||
|
||
class LatexError(Exception):
|
||
def __init__(self, msg, s, i):
|
||
ctx = s[max(0, i - 50):i + 50].replace('\n', ' ')
|
||
super().__init__('%s\n near: ...%s...' % (msg, ctx))
|
||
|
||
|
||
CHAPTER_MACROS = ('chapstart', 'chaphead', 'chapnum', 'partstart',
|
||
'sectionstart', 'matterstart')
|
||
|
||
|
||
def plain(tex):
|
||
"""A heading as plain text, for <title> and the navigation: markup
|
||
stripped, line breaks become a space."""
|
||
c = Converter()
|
||
h = c.convert(tex).replace('<br/>', ' ')
|
||
return html.unescape(re.sub(r'<[^>]+>', '', h)).strip()
|
||
|
||
|
||
def split_chapters(files):
|
||
"""Group page files into chapters at the heading macros."""
|
||
chapters = []
|
||
cur = {'title': 'Front matter', 'short': 'Front matter', 'kind': 'front',
|
||
'body': [], 'files': []}
|
||
for path in files:
|
||
# The polished text, not the raw transcription: tools/polish.py is the
|
||
# edition's typographic pass (range en dashes, ties, cross-reference
|
||
# links, unbreakable specimens), and the EPUB takes it from the same
|
||
# function the PDF build does, so a new rule reaches both editions.
|
||
raw = polish.polish(open(path).read())
|
||
pos = 0
|
||
while True:
|
||
m = re.search(r'\\(%s)' % '|'.join(CHAPTER_MACROS), raw[pos:])
|
||
if not m:
|
||
cur['body'].append(raw[pos:])
|
||
cur['files'].append(path)
|
||
break
|
||
start = pos + m.start()
|
||
cur['body'].append(raw[pos:start])
|
||
cur['files'].append(path)
|
||
j = start + m.end() - m.start()
|
||
cmd = m.group(1)
|
||
short = None
|
||
label = None
|
||
if cmd in ('chapstart', 'chaphead'):
|
||
short, j = opt_arg(raw, j)
|
||
title, j = balanced(raw, j)
|
||
elif cmd == 'chapnum':
|
||
a, j = balanced(raw, j)
|
||
b, j = balanced(raw, j)
|
||
# "Chapter 1" over "Charles Wilkins", as the print edition sets
|
||
# them -- joined by a line break, never by punctuation the
|
||
# original does not have (an earlier version invented ": ")
|
||
title = a + r'\\' + b
|
||
short = b
|
||
label = a
|
||
elif cmd == 'matterstart':
|
||
title, j = balanced(raw, j)
|
||
short = title
|
||
else: # partstart / sectionstart
|
||
a, j = balanced(raw, j)
|
||
b, j = balanced(raw, j)
|
||
title = a + r'\\' + b # was a + ' — ' + b: an invented dash
|
||
short = b
|
||
label = a
|
||
chapters.append(cur)
|
||
cur = {'title': title, 'short': short or title, 'kind': cmd,
|
||
'label': label, 'body': [], 'files': [path]}
|
||
pos = j
|
||
chapters.append(cur)
|
||
return [c for c in chapters if ''.join(c['body']).strip()]
|
||
|
||
|
||
ENV_RE = re.compile(r'\\begin\{(\w+)\}(.*?)\\end\{\1\}', re.S)
|
||
|
||
|
||
def extract_envs(conv, text, blocks):
|
||
"""Convert environments to HTML and stash them behind placeholders.
|
||
|
||
The generated markup must never go back through convert(), which escapes
|
||
< and >. An earlier version ran handle_envs() and then paragraphs() over
|
||
its output, so every <div>, <tr> and <blockquote> it had just produced was
|
||
re-escaped and surfaced as visible text in the reader. Placeholders keep
|
||
the two passes apart: the paragraph pass only ever sees LaTeX.
|
||
"""
|
||
def repl(m):
|
||
env, inner = m.group(1), m.group(2)
|
||
if env == 'tikzpicture':
|
||
if '\\sealbody' not in inner:
|
||
raise LatexError('a tikzpicture other than the version mark', inner, 0)
|
||
key = '@@BLOCK%d@@' % len(blocks)
|
||
blocks[key] = ('<div class="mark"><img src="images/mark.svg" '
|
||
'alt="Version mark, generated from %s" width="%d" height="%d"/></div>'
|
||
% (escape(SEAL['seed']), SEAL['w'], SEAL['h']))
|
||
return '\n\n' + key + '\n\n'
|
||
if env == 'tabular':
|
||
# the column spec contains nested braces (l@{}c@{}...), so a
|
||
# non-greedy {...} strip leaves "l@c@l@c@" glued to the first cell
|
||
k = 0
|
||
while k < len(inner) and inner[k] in ' \n\t':
|
||
k += 1
|
||
if k < len(inner) and inner[k] == '{':
|
||
_spec, k = balanced(inner, k)
|
||
inner = inner[k:]
|
||
rows = []
|
||
for line in inner.split('\\\\'):
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
cells = [conv.convert(c.strip()) for c in line.split('&')]
|
||
rows.append('<tr>' + ''.join('<td>%s</td>' % c for c in cells) + '</tr>')
|
||
body = '<table class="translit">' + ''.join(rows) + '</table>'
|
||
else:
|
||
inner = extract_envs(conv, inner, blocks)
|
||
if env == 'addmargin':
|
||
inner = re.sub(r'^\s*(\[[^\]]*\])?\s*(\{[^}]*\})?', '', inner)
|
||
body = paragraphs(conv, inner)
|
||
if env == 'extract':
|
||
body = '<blockquote>' + body + '</blockquote>'
|
||
elif env == 'biblist':
|
||
body = '<div class="biblist">' + body + '</div>'
|
||
elif env == 'center':
|
||
body = '<div class="center">' + body + '</div>'
|
||
elif env == 'addmargin':
|
||
body = '<div class="inset">' + body + '</div>'
|
||
key = '@@BLOCK%d@@' % len(blocks)
|
||
blocks[key] = body
|
||
return '\n\n' + key + '\n\n'
|
||
|
||
prev = None
|
||
while prev != text:
|
||
prev = text
|
||
text = ENV_RE.sub(repl, text)
|
||
return text
|
||
|
||
|
||
def restore(text, blocks):
|
||
prev = None
|
||
while prev != text:
|
||
prev = text
|
||
for k, v in blocks.items():
|
||
text = text.replace(k, v)
|
||
return text
|
||
|
||
|
||
BLOCK_START = ('<figure', '<blockquote', '<div', '<table', '<h2', '<h3', '<p class')
|
||
|
||
|
||
PAGEBREAK_ONLY = re.compile(r'(<span epub:type="pagebreak"[^>]*>[^<]*</span>\s*)+')
|
||
BLOCK_AFTER_MARKERS = re.compile(
|
||
r'(<span epub:type="pagebreak"[^>]*>[^<]*</span>\s*)+(%s)' % '|'.join(
|
||
re.escape(b) for b in ('<figure', '<blockquote', '<div', '<table',
|
||
'<h2', '<h3', '<p class', '@@BLOCK')))
|
||
|
||
|
||
def paragraphs(conv, text):
|
||
text = re.sub(r'(?m)^\s*%.*$', '', text)
|
||
out = []
|
||
pending = []
|
||
for para in re.split(r'\n\s*\n', text):
|
||
para = para.strip()
|
||
if not para:
|
||
continue
|
||
if re.fullmatch(r'@@BLOCK\d+@@', para):
|
||
# a stashed block: never re-convert it. Any page marker held back
|
||
# goes in front of it, so a break before a quotation stays before it.
|
||
out.append(''.join(pending) + para)
|
||
pending = []
|
||
continue
|
||
h = conv.convert(para).strip()
|
||
if not h:
|
||
continue
|
||
# A page break that falls between paragraphs arrives as a paragraph
|
||
# of its own. Wrapped in <p> it renders as a stray blank line (31 of
|
||
# them), so hold it and set it at the head of whatever follows.
|
||
if PAGEBREAK_ONLY.fullmatch(h):
|
||
pending.append(h)
|
||
continue
|
||
if pending:
|
||
h = ''.join(pending) + h
|
||
pending = []
|
||
if h.startswith(BLOCK_START) or h.startswith('<span epub:type="pagebreak"') \
|
||
and BLOCK_AFTER_MARKERS.match(h):
|
||
out.append(h)
|
||
else:
|
||
out.append('<p>' + h + '</p>')
|
||
out.extend(pending) # a trailing marker: keep, unwrapped
|
||
return '\n'.join(out)
|
||
|
||
|
||
XHTML = """<?xml version="1.0" encoding="utf-8"?>
|
||
<!DOCTYPE html>
|
||
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops"
|
||
lang="en" xml:lang="en">
|
||
<head><meta charset="utf-8"/><title>%s</title>
|
||
<link rel="stylesheet" type="text/css" href="style.css"/></head>
|
||
<body>
|
||
%s
|
||
</body></html>
|
||
"""
|
||
|
||
CSS = """/* The print edition's typography, carried wherever it survives reflow.
|
||
Source of each rule: src/preamble.tex. Disposition, rule by rule:
|
||
|
||
CARRIED
|
||
first-line indent 1.6em, 0.15em paragraph space -> as is
|
||
no indent after a heading or display -> as is
|
||
ragged right, no hyphenation (typescript breaks
|
||
only at spaces and typed hyphens) -> text-align:left, hyphens:none
|
||
widows and orphans forbidden -> widows/orphans 2
|
||
hanging punctuation -> hanging-punctuation (Apple
|
||
Books honours it; others ignore)
|
||
extract: indented both sides, set solid, no indent -> margins in em, tighter leading
|
||
bibliography: hanging indent 1.6em, set solid -> as is
|
||
cross-reference links: a hairline rule in linkink
|
||
(95,125,165), the text itself uncoloured -> underline, 1px, that colour
|
||
Contents / Plates whole-line links: unmarked -> as is
|
||
endnotes and plate captions a step smaller -> as is
|
||
specimens at native size (1px = 1/300in) -> height in em, per image
|
||
original page numbers in the outer margin, in sans -> floated to the right edge
|
||
of the line the page starts on
|
||
range en dashes, ties, specimen unbreakability -> from tools/polish.py itself
|
||
|
||
DROPPED, because reflow makes them meaningless or harmful
|
||
13pt XCharter on a 5.95in measure the reader sets face, size and margins
|
||
22.9pt absolute leading an absolute leading fights the reader's
|
||
own size; a relative 1.5 is kept instead
|
||
running heads there is no fixed page; readers show
|
||
their own, and the page-list carries
|
||
the original pagination for navigation
|
||
microtype expansion, raggedbottom no equivalent in CSS
|
||
*/
|
||
html { font-family: Georgia, "Times New Roman", serif; }
|
||
body { margin: 0 5%; line-height: 1.5; text-align: left;
|
||
hyphens: none; -webkit-hyphens: none; }
|
||
p { margin: 0 0 0.15em; text-indent: 1.6em; widows: 2; orphans: 2;
|
||
hanging-punctuation: first last; }
|
||
h1 + p, h2 + p, h3 + p, blockquote + p, figure + p, table + p, div + p,
|
||
.subhead + p, section > p:first-child { text-indent: 0; }
|
||
h1, h2, h3 { font-weight: bold; line-height: 1.25; margin: 1.2em 0 0.6em;
|
||
page-break-after: avoid; }
|
||
h1 { font-size: 1.5em; text-align: center; }
|
||
h2 { font-size: 1.25em; }
|
||
h3 { font-size: 1.1em; }
|
||
.subhead { font-style: italic; margin-top: 1em; text-indent: 0; }
|
||
.center { text-align: center; }
|
||
.center p { text-indent: 0; }
|
||
.inset { margin: 1em 1.5em; font-size: 0.9em; }
|
||
.inset p { text-indent: 0; }
|
||
|
||
/* extract: indented both sides (print 0.5in / 0.3in), set solid, unindented */
|
||
blockquote { margin: 0.8em 1.5em 0.8em 2.5em; line-height: 1.35; }
|
||
blockquote p { text-indent: 0; margin-bottom: 0.4em; }
|
||
|
||
/* bibliography: solid, continuation lines hang 1.6em */
|
||
.biblist p { text-indent: -1.6em; padding-left: 1.6em; margin: 0; }
|
||
h3.bibgroup { margin-top: 1em; }
|
||
h2.bibhead { text-align: center; }
|
||
|
||
/* links: a hairline rule in the print edition's linkink, text uncoloured */
|
||
a { color: inherit; }
|
||
a.xref { text-decoration: underline; text-decoration-thickness: 1px;
|
||
text-decoration-color: rgb(95,125,165); text-underline-offset: 0.18em; }
|
||
a.pgref, a.noteref { text-decoration: none; }
|
||
|
||
/* Images: width and height attributes on every <img> give the reader the real
|
||
aspect ratio; these rules let it scale without distorting -- the common EPUB
|
||
fault is a reader stretching an image to the frame. */
|
||
img { max-width: 100%; height: auto; }
|
||
figure.plate { margin: 1.2em 0; text-align: center; page-break-inside: avoid; }
|
||
figure.plate img { max-width: 100%; max-height: 88vh; width: auto; height: auto; }
|
||
figcaption { font-size: 0.85em; text-align: left; margin-top: 0.4em; }
|
||
.figure { text-align: center; margin: 1em 0; }
|
||
img.spec { vertical-align: -0.28em; width: auto; } /* height set per image */
|
||
.nowrap { white-space: nowrap; }
|
||
/* original page number: small sans at the right edge of the line the page
|
||
begins on, as the print edition sets it in the outer margin */
|
||
.pgnum { float: right; clear: right; margin: 0.25em 0 0 0.6em;
|
||
font-family: "Liberation Sans", Arial, sans-serif; font-size: 0.65em;
|
||
font-weight: normal; font-style: normal; line-height: 1;
|
||
text-indent: 0; color: #8a8a8a; } /* \\mbox: never broken */
|
||
|
||
/* Note numbers and other superscripts must not open up the line they sit on
|
||
-- the print edition's leading is even. A plain <sup> raises its baseline
|
||
and so enlarges the line box; lifting it by relative position instead
|
||
leaves the line box alone. */
|
||
sup { font-size: 0.7em; line-height: 0; vertical-align: baseline;
|
||
position: relative; top: -0.5em; }
|
||
|
||
/* Bengali in Tiro Bangla, embedded -- the print edition sets all Bengali in it */
|
||
@font-face { font-family: "Tiro Bangla"; font-style: normal; font-weight: normal;
|
||
src: url(fonts/TiroBangla-Regular.ttf); }
|
||
@font-face { font-family: "Tiro Bangla"; font-style: italic; font-weight: normal;
|
||
src: url(fonts/TiroBangla-Italic.ttf); }
|
||
.bn, :lang(bn) { font-family: "Tiro Bangla", serif; }
|
||
.bn { font-size: 1.15em; }
|
||
.bn .bn { font-size: 1em; } /* nested runs must not compound */
|
||
|
||
/* About this edition: set solid and unindented, a step smaller, as in print */
|
||
.colophon p { text-indent: 0; margin: 0 0 0.6em; font-size: 0.92em; }
|
||
.colophon .inset p, .colophon .inset { text-align: center; }
|
||
.mark { text-align: center; margin: 1em 0; }
|
||
/* the print mark stands about 2.5 text-heights tall; em keeps that proportion
|
||
at whatever size the reader sets, where its intrinsic 37px would not */
|
||
.mark img { height: 2.6em; width: auto; max-width: 100%; }
|
||
img.inline-graphic { max-height: 4em; width: auto; }
|
||
|
||
/* Errata: the original's page in a column of its own, the entry hanging */
|
||
.errata p { text-indent: 0; }
|
||
p.erratum { padding-left: 4.5em; text-indent: -4.5em; margin: 0 0 0.35em; }
|
||
p.erratum a.pgref { display: inline-block; min-width: 4.5em; text-indent: 0; }
|
||
|
||
/* Notes: one section at the back, grouped by chapter, a step smaller */
|
||
.notegroup h2 { font-size: 1.05em; margin: 1.4em 0 0.5em; }
|
||
div.note p { text-indent: 0; margin: 0 0 0.35em; font-size: 0.9em; }
|
||
a.noteback { text-decoration: none; }
|
||
.smallcaps { font-variant: small-caps; }
|
||
table.translit { border-collapse: collapse; margin: 1em 0; }
|
||
table.translit td { padding: 0.15em 0.6em 0.15em 0; vertical-align: baseline; }
|
||
.plentry, .tocl { margin: 0 0 0.3em; padding-left: 2em; text-indent: -2em; }
|
||
.plnum { display: inline-block; min-width: 2.2em; }
|
||
.unsure { border-bottom: 1px dotted #999; }
|
||
.notes { margin-top: 2em; border-top: 1px solid #ccc; padding-top: 1em;
|
||
font-size: 0.9em; }
|
||
.notes h2 { font-size: 1.1em; }
|
||
.notes p { text-indent: 0; }
|
||
aside.note { margin: 0 0 0.5em; }
|
||
"""
|
||
|
||
|
||
def finish_documents(names):
|
||
r"""Whole-book passes over the written chapters.
|
||
|
||
1. Fragment links. Anything the converter links by id -- an original page
|
||
(\pg, \xref, the Contents and List of Plates), a plate, a note -- is
|
||
emitted as a same-document "#id", because at conversion time it cannot
|
||
know which chapter file the target lands in. Resolve every one against
|
||
an index of all ids in the book; one that resolves nowhere fails the
|
||
build. (199 dead Contents and Plates links came from resolving only the
|
||
targets that happened to share a file.)
|
||
|
||
2. Empty blocks. A block holding nothing -- or only page-break markers --
|
||
renders as a stray blank line. Unwrap the markers (they stay, bare, in
|
||
the flow) and drop the empty block. Applies to every block element,
|
||
not only the <p> where it was first noticed.
|
||
"""
|
||
docs = {n: open(os.path.join(OEBPS, n)).read() for n in names}
|
||
owner = {}
|
||
for n, d in docs.items():
|
||
for i in re.findall(r'\bid="([^"]+)"', d):
|
||
owner.setdefault(i, n)
|
||
marker = r'<span epub:type="pagebreak"[^>]*>[^<]*</span>'
|
||
empty = re.compile(r'<(p|div|blockquote|section|li|figcaption)(\s[^>]*)?>'
|
||
r'((?:\s|%s)*)</\1>' % marker)
|
||
for n, d in docs.items():
|
||
def fix(m, n=n):
|
||
frag = m.group(1)
|
||
if re.search(r'\bid="%s"' % re.escape(frag), docs[n]):
|
||
return m.group(0)
|
||
if frag not in owner:
|
||
raise SystemExit('make_epub: link to #%s in %s resolves nowhere' % (frag, n))
|
||
return 'href="%s#%s"' % (owner[frag], frag)
|
||
d = re.sub(r'href="#([^"]+)"', fix, d)
|
||
prev = None
|
||
while prev != d:
|
||
prev = d
|
||
d = empty.sub(lambda m: m.group(3).strip(), d)
|
||
open(os.path.join(OEBPS, n), 'w').write(d)
|
||
|
||
|
||
ROMAN = ['I', 'II', 'III', 'IV', 'V', 'VI', 'VII', 'VIII', 'IX', 'X']
|
||
|
||
|
||
def nav_items(heads):
|
||
"""(level, label, file) for each chapter, as the PDF outline gives them.
|
||
|
||
The print edition's outline nests Parts > Sections > Chapters and labels
|
||
them by position -- I, I.A, I.A.1; II.8 where a Part has no Sections --
|
||
keeping the author's chapter numbers (src/preamble.tex, "navigation
|
||
numbering"). The same labels head the PDF's note groups ("Notes to I.A.1
|
||
Charles Wilkins"), so both the outline and the notes take them from here.
|
||
"""
|
||
items = []
|
||
part = sec = 0
|
||
for fn, ch in heads:
|
||
k = ch['kind']
|
||
name = plain(ch['short'])
|
||
if k == 'front':
|
||
continue
|
||
if k == 'partstart':
|
||
part += 1; sec = 0
|
||
items.append((1, '%s %s' % (ROMAN[part - 1], name), fn))
|
||
elif k == 'sectionstart':
|
||
sec += 1
|
||
items.append((2, '%s.%s %s' % (ROMAN[part - 1], chr(64 + sec), name), fn))
|
||
elif k == 'chapnum':
|
||
num = re.search(r'\d+', ch['label'] or '').group(0)
|
||
pos = ROMAN[part - 1] + ('.' + chr(64 + sec) if sec else '')
|
||
items.append((3 if sec else 2, '%s.%s %s' % (pos, num, name), fn))
|
||
else: # unnumbered: top level
|
||
items.append((1, name, fn))
|
||
return items
|
||
|
||
|
||
def outline(items):
|
||
return _nest(items, 0, 1)[0]
|
||
|
||
|
||
def _nest(items, i, level):
|
||
"""items[i:] as <li> elements at `level`, deeper items nested inside the
|
||
<li> that precedes them. Returns (html, next index)."""
|
||
out = []
|
||
while i < len(items) and items[i][0] >= level:
|
||
_lvl, text, fn = items[i]
|
||
out.append('<li><a href="%s">%s</a>' % (fn, escape(text)))
|
||
i += 1
|
||
if i < len(items) and items[i][0] > level:
|
||
inner, i = _nest(items, i, level + 1)
|
||
out.append('<ol>' + inner + '</ol>')
|
||
out.append('</li>')
|
||
return ''.join(out), i
|
||
|
||
|
||
def build_mark():
|
||
r"""Render the version mark to SVG, from tools/gen_seal.py itself.
|
||
|
||
The PDF draws the mark with TikZ from a hash of the sources; the EPUB takes
|
||
the same seed and the same drawing code and renders that TikZ to SVG
|
||
(XeLaTeX, then pdftocairo). Vector, so it holds at any size, and the same
|
||
mark as the PDF's whenever both are built from the same sources.
|
||
"""
|
||
import gen_seal
|
||
seed = gen_seal.content_hash()
|
||
body = gen_seal.build(seed)
|
||
tmp = os.path.join(BUILD, 'mark')
|
||
os.makedirs(tmp, exist_ok=True)
|
||
with open(os.path.join(tmp, 'mark.tex'), 'w') as f:
|
||
f.write('\\documentclass[tikz,border=2pt]{standalone}\n\\usepackage{xcolor}\n'
|
||
'\\definecolor{ink}{RGB}{26,26,28}\\definecolor{paper}{RGB}{255,255,255}\n'
|
||
'\\begin{document}\\begin{tikzpicture}[scale=0.62]\n%s\n'
|
||
'\\end{tikzpicture}\\end{document}\n' % body)
|
||
subprocess.run(['xelatex', '-interaction=nonstopmode', 'mark.tex'], cwd=tmp,
|
||
check=True, stdout=subprocess.DEVNULL)
|
||
svg = os.path.join(IMG, 'mark.svg')
|
||
subprocess.run(['pdftocairo', '-svg', os.path.join(tmp, 'mark.pdf'), svg], check=True)
|
||
m = re.search(r'<svg[^>]*\bwidth="([\d.]+)(?:pt)?"[^>]*\bheight="([\d.]+)(?:pt)?"',
|
||
open(svg).read())
|
||
w, h = (float(m.group(1)), float(m.group(2))) if m else (60.0, 60.0)
|
||
return {'seed': seed, 'w': int(round(w)), 'h': int(round(h))}
|
||
|
||
|
||
SEAL = {'seed': '', 'w': 0, 'h': 0}
|
||
|
||
|
||
def build_cover():
|
||
"""Render the PDF's cover page to a raster cover image.
|
||
|
||
The EPUB's cover is the print cover, not a re-layout of it: the mosaic's
|
||
bleed and the title's proportions only hold at the designed trim, so it is
|
||
reproduced as one image rather than reflowed.
|
||
"""
|
||
src = os.path.join(ROOT, 'ross-1988-retypeset.pdf')
|
||
if not os.path.exists(src):
|
||
return None
|
||
r = subprocess.run(['pdftoppm', '-png', '-r', '170', '-f', '1', '-l', '1', src],
|
||
capture_output=True)
|
||
if not r.stdout:
|
||
return None
|
||
import io
|
||
im = Image.open(io.BytesIO(r.stdout)).convert('RGB')
|
||
dst = os.path.join(IMG, 'cover.jpg')
|
||
im.save(dst, 'JPEG', quality=88, optimize=True)
|
||
return im.size
|
||
|
||
|
||
def convert_matter(conv, path, fname):
|
||
"""Convert one of the PDF's own front- or back-matter files (src/colophon.tex,
|
||
src/errata.tex) exactly as the pages are converted. They are read raw, as
|
||
build.sh reads them: polish.py is a pass over src/pages only."""
|
||
conv.cur_file = fname
|
||
blocks = {}
|
||
body = extract_envs(conv, open(os.path.join(ROOT, path)).read(), blocks)
|
||
return restore(paragraphs(conv, body), blocks)
|
||
|
||
|
||
def write_doc(fname, title, body):
|
||
with open(os.path.join(OEBPS, fname), 'w') as f:
|
||
f.write(XHTML % (escape(title), body))
|
||
|
||
|
||
def main():
|
||
if os.path.exists(BUILD):
|
||
shutil.rmtree(BUILD)
|
||
os.makedirs(IMG, exist_ok=True)
|
||
os.makedirs(os.path.join(OEBPS, 'fonts'), exist_ok=True)
|
||
SEAL.update(build_mark()) # the colophon's handlers read it
|
||
|
||
files = sorted(glob.glob(os.path.join(ROOT, 'src', 'pages', 'p*.tex')))
|
||
chapters = split_chapters(files)
|
||
|
||
conv = Converter()
|
||
heads, chdocs, notes_by = [], [], []
|
||
for idx, ch in enumerate(chapters):
|
||
fname = 'ch%03d.xhtml' % idx
|
||
conv.cur_file = fname
|
||
conv.notes = []
|
||
blocks = {}
|
||
body = extract_envs(conv, ''.join(ch['body']), blocks)
|
||
body = restore(paragraphs(conv, body), blocks)
|
||
# Headings go through the same converter as the text, so a two-line
|
||
# title keeps its break. The title page is its own heading, so none is
|
||
# invented for it (an earlier version printed "Front matter" there).
|
||
if ch['kind'] == 'front':
|
||
head = ''
|
||
# epub:type only: DPUB-ARIA has no title-page role, and doc-cover
|
||
# would announce this page to a screen reader as the cover image
|
||
body = '<section epub:type="titlepage">' + body + '</section>'
|
||
else:
|
||
head = '<h1>%s</h1>' % conv.convert(ch['title'])
|
||
write_doc(fname, plain(ch['short']), head + '\n' + body)
|
||
chdocs.append(fname)
|
||
heads.append((fname, ch))
|
||
if conv.notes:
|
||
notes_by.append((fname, list(conv.notes)))
|
||
|
||
items = nav_items(heads)
|
||
label = {fn: text for _lvl, text, fn in items}
|
||
|
||
# ---- About this edition: the PDF's own colophon, the EPUB's wording
|
||
conv.notes = []
|
||
write_doc('colophon.xhtml', 'About this edition',
|
||
'<section class="colophon" epub:type="colophon">%s</section>'
|
||
% convert_matter(conv, 'src/colophon.tex', 'colophon.xhtml'))
|
||
|
||
# ---- Notes: one section at the back, grouped as the PDF groups them
|
||
# ("Notes to I.A.1 Charles Wilkins"), each note linked to its reference and
|
||
# back. The PDF sets all its endnotes after the text; so does the EPUB now.
|
||
groups = []
|
||
for fn, notes in notes_by:
|
||
if fn not in label:
|
||
raise SystemExit('make_epub: notes in %s, which has no outline label' % fn)
|
||
groups.append(
|
||
'<section class="notegroup"><h2>Notes to %s</h2>%s</section>'
|
||
% (escape(label[fn]), ''.join(
|
||
'<div class="note" epub:type="endnote" id="%s"><p>'
|
||
'<a class="noteback" role="doc-backlink" href="#%sr">%s.</a> %s</p></div>'
|
||
% (nid, nid, num, txt) for num, nid, txt in notes)))
|
||
write_doc('notes.xhtml', 'Notes',
|
||
'<section epub:type="endnotes" role="doc-endnotes"><h1>Notes</h1>%s</section>'
|
||
% ''.join(groups))
|
||
|
||
# ---- Errata: src/errata.tex, the same list the PDF sets
|
||
write_doc('errata.xhtml', 'Errata',
|
||
'<section class="errata" epub:type="errata"><h1>Errata</h1>%s</section>'
|
||
% convert_matter(conv, 'src/errata.tex', 'errata.xhtml'))
|
||
|
||
order = [chdocs[0], 'colophon.xhtml'] + chdocs[1:] + ['notes.xhtml', 'errata.xhtml']
|
||
finish_documents(order)
|
||
|
||
cover_size = build_cover()
|
||
if cover_size:
|
||
cw, chh = cover_size
|
||
write_doc('cover.xhtml', 'Cover',
|
||
'<div class="center" style="margin:0;padding:0">'
|
||
'<img src="images/cover.jpg" alt="%s" width="%d" height="%d"/>'
|
||
'</div>' % (escape(TITLE), cw, chh))
|
||
|
||
# Tiro Bangla, as the PDF sets all Bengali in it. SIL OFL 1.1 (its licence
|
||
# travels in the font's own name table, which the OFL accepts); fsType 0,
|
||
# so embedding is unrestricted.
|
||
for face in ('TiroBangla-Regular.ttf', 'TiroBangla-Italic.ttf'):
|
||
shutil.copy(os.path.join(ROOT, 'fonts', face), os.path.join(OEBPS, 'fonts', face))
|
||
|
||
with open(os.path.join(OEBPS, 'style.css'), 'w') as f:
|
||
f.write(CSS)
|
||
|
||
# ---- navigation: the PDF's outline, and the page-list
|
||
toc = outline([(1, 'About this edition', 'colophon.xhtml')] + items +
|
||
[(1, 'Notes', 'notes.xhtml'), (1, 'Errata', 'errata.xhtml')])
|
||
seen = set()
|
||
plist = []
|
||
for num, fn, anchor in conv.pages:
|
||
if num in seen:
|
||
continue
|
||
seen.add(num)
|
||
plist.append('<li><a href="%s#%s">%s</a></li>' % (fn, anchor, num))
|
||
write_doc('nav.xhtml', 'Contents', """
|
||
<nav epub:type="toc" role="doc-toc" id="toc"><h1>Contents</h1><ol>%s</ol></nav>
|
||
<nav epub:type="landmarks" hidden=""><ol>
|
||
<li><a epub:type="cover" href="cover.xhtml">Cover</a></li>
|
||
<li><a epub:type="titlepage" href="%s">Title page</a></li>
|
||
<li><a epub:type="bodymatter" href="%s">Beginning</a></li>
|
||
<li><a epub:type="endnotes" href="notes.xhtml">Notes</a></li>
|
||
</ol></nav>
|
||
<nav epub:type="page-list" role="doc-pagelist" hidden=""><ol>%s</ol></nav>
|
||
""" % (toc, chdocs[0], chdocs[1], ''.join(plist)))
|
||
docs = [(fn, '') for fn in order]
|
||
|
||
# ---- package
|
||
items, spine = [], []
|
||
if cover_size:
|
||
items.append('<item id="cover" href="cover.xhtml" media-type="application/xhtml+xml"/>')
|
||
items.append('<item id="cover-image" href="images/cover.jpg" '
|
||
'media-type="image/jpeg" properties="cover-image"/>')
|
||
spine.append('<itemref idref="cover"/>')
|
||
items.append('<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" '
|
||
'properties="nav"/>')
|
||
items.append('<item id="css" href="style.css" media-type="text/css"/>')
|
||
for i, (fn, _t) in enumerate(docs):
|
||
items.append('<item id="c%d" href="%s" media-type="application/xhtml+xml"/>' % (i, fn))
|
||
spine.append('<itemref idref="c%d"/>' % i)
|
||
MT = {'.jpg': 'image/jpeg', '.png': 'image/png', '.svg': 'image/svg+xml'}
|
||
for f in sorted(os.listdir(IMG)):
|
||
if f == 'cover.jpg':
|
||
continue
|
||
items.append('<item id="img-%s" href="images/%s" media-type="%s"/>'
|
||
% (re.sub(r'\W', '_', f), f, MT[os.path.splitext(f)[1]]))
|
||
for f in sorted(os.listdir(os.path.join(OEBPS, 'fonts'))):
|
||
items.append('<item id="font-%s" href="fonts/%s" media-type="font/ttf"/>'
|
||
% (re.sub(r'\W', '_', f), f))
|
||
opf = """<?xml version="1.0" encoding="utf-8"?>
|
||
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="uid"
|
||
xml:lang="en" prefix="dcterms: http://purl.org/dc/terms/">
|
||
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
|
||
<dc:identifier id="uid">%s</dc:identifier>
|
||
<dc:title>%s</dc:title>
|
||
<dc:creator>%s</dc:creator>
|
||
<dc:language>en</dc:language>
|
||
<dc:date>2026</dc:date>
|
||
<dc:publisher>Re-typeset edition</dc:publisher>
|
||
<dc:subject>Bengali type</dc:subject>
|
||
<dc:subject>Typography</dc:subject>
|
||
<dc:subject>Printing history</dc:subject>
|
||
<dc:description>A re-typeset, searchable edition of the 1988 SOAS Ph.D. thesis, made from the ProQuest scan (number 10731406). The original pagination is preserved as EPUB page-list markers.</dc:description>
|
||
<!-- The page numbers in this book are those of the 1988 typescript; this
|
||
names the edition they come from, which is what lets a reader report
|
||
"page 214" and mean the same place another copy does. -->
|
||
<dc:source id="src">urn:proquest:10731406</dc:source>
|
||
<meta refines="#src" property="dcterms:modified">2026-01-01T00:00:00Z</meta>
|
||
<meta property="dcterms:modified">2026-01-01T00:00:00Z</meta>
|
||
<meta property="pageBreakSource" refines="#src">Ph.D. thesis, SOAS, University of London, 1988</meta>
|
||
<meta property="schema:accessMode">textual</meta>
|
||
<meta property="schema:accessMode">visual</meta>
|
||
<meta property="schema:accessibilityFeature">printPageNumbers</meta>
|
||
<meta property="schema:accessibilityFeature">tableOfContents</meta>
|
||
<meta property="schema:accessibilityFeature">longDescription</meta>
|
||
</metadata>
|
||
<manifest>%s</manifest>
|
||
<spine>%s</spine>
|
||
</package>
|
||
""" % (UID, escape(TITLE), escape(AUTHOR), ''.join(items), ''.join(spine))
|
||
with open(os.path.join(OEBPS, 'content.opf'), 'w') as f:
|
||
f.write(opf)
|
||
|
||
os.makedirs(os.path.join(BUILD, 'META-INF'), exist_ok=True)
|
||
with open(os.path.join(BUILD, 'META-INF', 'container.xml'), 'w') as f:
|
||
f.write('<?xml version="1.0" encoding="utf-8"?>\n'
|
||
'<container version="1.0" '
|
||
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
|
||
'<rootfiles><rootfile full-path="OEBPS/content.opf" '
|
||
'media-type="application/oebps-package+xml"/></rootfiles></container>\n')
|
||
with open(os.path.join(BUILD, 'mimetype'), 'w') as f:
|
||
f.write('application/epub+zip')
|
||
|
||
# ---- zip: mimetype first and stored, as the spec requires
|
||
if os.path.exists(OUT):
|
||
os.remove(OUT)
|
||
with zipfile.ZipFile(OUT, 'w') as z:
|
||
z.write(os.path.join(BUILD, 'mimetype'), 'mimetype', zipfile.ZIP_STORED)
|
||
for base, _dirs, fs in os.walk(BUILD):
|
||
for f in sorted(fs):
|
||
if f == 'mimetype':
|
||
continue
|
||
full = os.path.join(base, f)
|
||
z.write(full, os.path.relpath(full, BUILD), zipfile.ZIP_DEFLATED)
|
||
|
||
print('documents : %d (%d chapters + colophon, notes, errata)' % (len(docs), len(chdocs)))
|
||
print('notes : %d, in %d groups' % (sum(len(n) for _f, n in notes_by), len(notes_by)))
|
||
print('page markers : %d unique original pages' % len(plist))
|
||
print('images : %d' % len(os.listdir(IMG)))
|
||
print('epub : %s (%.1f MB)' % (os.path.relpath(OUT, ROOT),
|
||
os.path.getsize(OUT) / 1048576))
|
||
|
||
|
||
if __name__ == '__main__':
|
||
sys.exit(main())
|