Files
bdeshiandClaude Sonnet 5.5 4424cc9887 Add the EPUB edition: converter, class-based audit, shared front matter
tools/make_epub.py converts the same src/ the PDF is set from, running
polish.py first, with original page numbers as page-list metadata, endnotes
gathered in one linked Notes section, and the colophon and errata included.
It raises on any macro it does not declare. tools/epub_audit.py checks by
class of fault (LaTeX residue, escaping, empty blocks, links, images, XML,
content, typography drift from the PDF); make epub runs both.

\byedition{PDF}{EPUB} lets the colophon carry the sentences that are true of
only one edition; the errata introduction moves into src/errata.tex so both
editions print one copy. reprocheck.py now accepts an .epub and skips
environment parameters that are layout, not copy.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
2026-09-29 14:36:18 +06:00

1245 lines
52 KiB
Python
Raw Permalink Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
r"""Build a reflowable EPUB 3 from the same src/pages transcription as the PDF.
docker compose run --rm -T tex python3 tools/make_epub.py
WHAT TRANSFERS AND WHAT DOES NOT
The PDF's typography is fixed-measure work -- a 22.9pt leading anchored to the
inline specimens, a 5.95in measure, hanging punctuation, ragged-right with
hyphenation off, plates scaled to the measure. None of that survives reflow and
none of it is attempted here. What the EPUB keeps is the edition's substance:
the text, the original pagination, the plates and inline specimens, the
endnotes, and the editorial apparatus.
PAGE NUMBERS
The 1988 pagination is the spine of this edition -- the Contents, the List of
Plates and the author's own cross-references all cite it. Every \origpage{N}
becomes an EPUB 3 pagebreak marker, and all of them are collected into the
page-list of the navigation document, which is what makes a reader's "go to
page" and page-number display work on a reflowable book. dc:source and
pageBreakSource name the print original the numbering belongs to.
IMAGES
Every <img> carries explicit width and height attributes taken from the file's
real pixel dimensions, so a reader that lays out before the image loads still
reserves the right box. The CSS sets width:auto/height:auto against those, which
is what stops the common EPUB failure of images stretched to the frame.
Inline type specimens keep their relative sizes: \ig sets them at 1px = 1/300in
in print, so here each gets a height in em derived from the same ratio -- a
vowel sign must not come out as tall as a letter.
"""
import glob
import html
import os
import re
import shutil
import subprocess
import sys
import unicodedata
import zipfile
from xml.sax.saxutils import escape
import numpy as np
from PIL import Image
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
# polish.py builds its plate/chapter maps from the relative path src/pages at
# import time, so it must be imported from the project root.
os.chdir(ROOT)
sys.path.insert(0, os.path.join(ROOT, 'tools'))
import polish # noqa: E402
BUILD = os.path.join(ROOT, 'work', 'epub')
OEBPS = os.path.join(BUILD, 'OEBPS')
IMG = os.path.join(OEBPS, 'images')
OUT = os.path.join(ROOT, 'ross-1988-retypeset.epub')
TITLE = 'The Evolution of the Printed Bengali Character from 1778 to 1978'
AUTHOR = 'Fiona Georgina Elisabeth Ross'
UID = 'urn:uuid:ross-1988-bengali-retypeset-2026'
# ---------------------------------------------------------------- helpers
def balanced(s, i):
"""Return (content, index after closing brace) for a { } group at s[i]=='{'."""
assert s[i] == '{'
depth = 0
j = i
while j < len(s):
if s[j] == '{':
depth += 1
elif s[j] == '}':
depth -= 1
if depth == 0:
return s[i + 1:j], j + 1
j += 1
return s[i + 1:], len(s)
def opt_arg(s, i):
"""Optional [..] argument."""
if i < len(s) and s[i] == '[':
j = s.index(']', i)
return s[i + 1:j], j + 1
return None, i
ESCAPED = {'&': '&amp;', '%': '%', '$': '$', '#': '#', '_': '_',
' ': ' ', ',': '\u2009', '{': '{', '}': '}'}
# Letter-name macros that stand for text. Without these they fall through to
# the "unknown macro" path and are dropped silently -- \ldots alone accounts
# for 118 ellipses in this text.
TEXT_MACROS = {
'ldots': '\u2026', 'dots': '\u2026',
'textasciitilde': '~', 'textcopyright': '\u00a9',
'textemdash': '\u2014', 'textendash': '\u2013',
'textquoteleft': '\u2018', 'textquoteright': '\u2019',
'textquotedblleft': '\u201c', 'textquotedblright': '\u201d',
'pounds': '\u00a3', 'dag': '\u2020', 'ddag': '\u2021',
'LaTeX': 'LaTeX', 'TeX': 'TeX', 'XeLaTeX': 'XeLaTeX',
}
def _glyph_table():
r"""Glyph index -> character for Tiro Bangla, read off the font's own cmap.
The transcription sets some isolated signs by raw glyph index. Deriving the
table from the font, rather than typing in the three indices the table uses
today, means any \XeTeXglyph the source ever uses resolves -- or fails the
build if the font has no character for it.
"""
from fontTools.ttLib import TTFont
font = TTFont(os.path.join(ROOT, 'fonts', 'TiroBangla-Regular.ttf'))
order = font.getGlyphOrder()
first = {}
for cp, g in sorted(font.getBestCmap().items()):
first.setdefault(g, cp)
return {gid: chr(first[g]) for gid, g in enumerate(order) if g in first}
GLYPH_CHARS = _glyph_table()
IMG_CACHE = {}
def img_size(path):
if path not in IMG_CACHE:
with Image.open(os.path.join(ROOT, path)) as im:
IMG_CACHE[path] = im.size
return IMG_CACHE[path]
def epub_image(src_rel):
"""Copy/convert an image into the EPUB and return (name, w, h).
Bilevel plates go to 1-bit PNG and halftone plates to JPEG, the same split
the PDF uses -- an 8-bit greyscale PNG of a page of type is mostly storing
scanner grain.
"""
name = src_rel.replace('/', '_')
dst = os.path.join(IMG, name)
if os.path.exists(dst):
return name, *img_size(src_rel)
a = np.array(Image.open(os.path.join(ROOT, src_rel)).convert('L'))
h = np.bincount(a.ravel(), minlength=256)
mid = 1.0 - (h[:40].sum() + h[216:].sum()) / a.size
if src_rel.startswith('plates/inline/'):
Image.fromarray(a).save(dst, 'PNG', optimize=True)
elif mid > 0.25:
name = os.path.splitext(name)[0] + '.jpg'
dst = os.path.join(IMG, name)
Image.fromarray(a).save(dst, 'JPEG', quality=85, optimize=True)
else:
Image.fromarray(a > 128).save(dst, 'PNG', optimize=True, bits=1)
w, hh = a.shape[1], a.shape[0]
IMG_CACHE[src_rel] = (w, hh)
return name, w, hh
# ---------------------------------------------------------------- text
def textify(s):
"""LaTeX text conventions -> HTML entities and Unicode."""
s = s.replace('\\&', '&amp;').replace('\\%', '%').replace('\\$', '$')
s = s.replace('\\#', '#').replace('\\_', '_')
s = s.replace('\\textasciitilde', '~')
s = s.replace('\\ldots\\ ', '… ').replace('\\ldots', '…')
s = s.replace('\\textcopyright{}', '©').replace('\\textcopyright', '©')
s = s.replace('\\,', '\u2009').replace('\\ ', ' ')
s = re.sub(r'``', '\u201c', s)
s = re.sub(r"''", '\u201d', s)
s = re.sub(r'`', '\u2018', s)
s = re.sub(r"(?<=[\w.,;:!?\u2019])'", '\u2019', s)
s = s.replace('---', '\u2014').replace('--', '\u2013')
s = s.replace('~', '\u00a0')
return s
def pagebreak(anchor, num):
"""An original-page marker, with the number as visible text.
The edition's whole reference apparatus -- the Contents, the List of
Plates, the author's own cross-references -- cites the 1988 pagination,
and the print edition sets each page's number in the outer margin so a
reader can see where a cited page begins. An empty marker carried the
numbers to the page-list, which serves navigation, but never to the page
itself; reprocheck counted all 128 of them missing. The number is floated
to the right edge of the line the page begins on -- reflow's equivalent of
the outer margin.
"""
return ('<span epub:type="pagebreak" role="doc-pagebreak" id="%s" '
'aria-label="%s" class="pgnum">%s</span>' % (anchor, num, num))
class Converter:
def __init__(self):
self.notes = [] # (n, html) for the current chapter
self.pages = [] # (page number, chapter file, anchor) in order
self.cur_file = None
def convert(self, s):
r"""LaTeX -> XHTML for one run of text.
Every control sequence is dispatched by a DECLARED signature and
consumes exactly the arguments that signature names -- no more, no
fewer. Anything undeclared stops the build (LatexError) instead of
being dropped. The earlier converter guessed: it dropped unknown macros
silently (118 ellipses went that way), and a generic "eat a following
[..]" rule would have deleted the editorial "[with]" that follows
\ldots in plate 6's caption. Signatures make both impossible.
"""
out = []
i = 0
n = len(s)
while i < n:
c = s[i]
if c == '\\':
m = re.match(r'\\([a-zA-Z]+)(\*?)', s[i:])
if not m: # control symbol
sym = s[i + 1:i + 2]
if sym == '\\': # \\ *? [length]?
out.append('<br/>')
k = i + 2
if s[k:k + 1] == '*':
k += 1
m2 = re.match(r'[ \t]*\[[^\]]*\]', s[k:])
if m2 and LENGTH.fullmatch(m2.group(0).strip()[1:-1].strip()):
k += m2.end()
i = k
continue
if sym in ESCAPED:
out.append(ESCAPED[sym])
i += 2
continue
raise LatexError('undeclared control symbol \\%s' % sym, s, i)
cmd = m.group(1)
j = i + m.end()
handler = getattr(self, 'cmd_' + cmd, None)
if handler:
txt, j = handler(s, j)
out.append(txt)
i = j
continue
if cmd in TEXT_MACROS:
out.append(TEXT_MACROS[cmd])
if s[j:j + 2] == '\\ ': # \ldots\ -- a real space
out.append(' ')
j += 2
i = j
continue
if cmd in DECLARATIONS:
# A declaration styles the rest of its group. convert() is
# called per group, so "the rest of the group" is the rest
# of s -- wherever in the group the switch falls, not only
# when it happens to come first.
rest = self.convert(s[j:])
tag = DECLARATIONS[cmd]
if tag:
rest = rest.lstrip()
rest = ('<span class="%s">%s</span>' % (tag[1:], rest)
if tag.startswith('.') else
'<%s>%s</%s>' % (tag, rest, tag))
out.append(rest)
i = n
continue
if cmd in LAYOUT:
nopt, nman = LAYOUT[cmd]
for _ in range(nopt):
_o, j = opt_arg(s, j)
for _ in range(nman):
_a, j = balanced(s, j)
i = j
continue
raise LatexError('undeclared macro \\%s' % cmd, s, i)
elif c == '{':
grp, j = balanced(s, i)
out.append(self.convert(grp))
i = j
elif c == '}':
i += 1
else:
k = i
while k < n and s[k] not in '\\{}':
k += 1
out.append(BENGALI_RUN.sub(
r'<span class="bn" lang="bn">\g<0></span>', escape(textify(s[i:k]))))
i = k
return ''.join(out)
# ---- macros
def cmd_newcommand(self, s, i):
"""Swallow a local macro definition. p0015/p0016 define \\B for the
transliteration table; without this the definition's own body is
rendered as if it were text."""
_name, i = balanced(s, i)
_n, i = opt_arg(s, i)
_body, i = balanced(s, i)
return '', i
cmd_renewcommand = cmd_newcommand
def cmd_xref(self, s, i):
r'''polish.py's cross-reference: \xref{page.N}{shown}. The target is an
original page, whose anchor is resolved across files after assembly.'''
target, i = balanced(s, i)
shown, i = balanced(s, i)
m = re.fullmatch(r'\s*page\.(\d+)\s*', target)
if not m:
raise LatexError('xref to unknown target %r' % target, s, i)
return '<a class="xref" href="#pg-%s">%s</a>' % (m.group(1), self.convert(shown)), i
cmd_hyperlink = cmd_xref
def cmd_mbox(self, s, i):
r'''\mbox keeps its content on one line. polish.py wraps every inline
specimen in one so it never parts from adjacent punctuation; on a
narrow reflowed screen that matters more, not less. Its content is
text -- the old drop-list would have discarded all 411 specimens.'''
a, i = balanced(s, i)
return '<span class="nowrap">' + self.convert(a) + '</span>', i
def cmd_byedition(self, s, i):
"""\byedition{PDF wording}{EPUB wording}: take the EPUB's."""
_pdf, i = balanced(s, i)
epub, i = balanced(s, i)
return self.convert(epub), i
def cmd_sealseed(self, s, i):
return escape(SEAL['seed']), i
def cmd_includegraphics(self, s, i):
_opt, i = opt_arg(s, i)
path, i = balanced(s, i)
path = path.strip()
name = os.path.basename(path)
dst = os.path.join(IMG, name)
if not os.path.exists(dst):
shutil.copy(os.path.join(ROOT, path), dst)
w, h = img_size(path)
return ('<img class="inline-graphic" src="images/%s" alt="" width="%d" '
'height="%d"/>' % (name, w, h)), i
def cmd_erratumline(self, s, i):
pg, i = balanced(s, i)
typed, i = balanced(s, i)
corrected, i = balanced(s, i)
pg = pg.strip()
return ('<p class="erratum"><a class="pgref" href="#pg-%s">p. %s</a> '
'reads ‘%s’; corrected here to ‘%s’</p>'
% (pg, pg, self.convert(typed), self.convert(corrected))), i
def cmd_texttt(self, s, i):
a, i = balanced(s, i)
return '<code>' + self.convert(a) + '</code>', i
def cmd_emph(self, s, i):
a, i = balanced(s, i)
return '<em>' + self.convert(a) + '</em>', i
def cmd_textbf(self, s, i):
a, i = balanced(s, i)
return '<strong>' + self.convert(a) + '</strong>', i
def cmd_textsuperscript(self, s, i):
a, i = balanced(s, i)
return '<sup>' + self.convert(a) + '</sup>', i
def cmd_B(self, s, i):
a, i = balanced(s, i)
return '<span class="bn">' + self.convert(a) + '</span>', i
def cmd_origpage(self, s, i):
a, i = balanced(s, i)
num = a.strip()
anchor = 'pg-%s' % num
self.pages.append((num, self.cur_file, anchor))
return pagebreak(anchor, num), i
def cmd_fn(self, s, i):
num, i = balanced(s, i)
body, i = balanced(s, i)
num = num.strip()
# Unique across the whole book, not just the chapter: every note now
# lands in one notes.xhtml, where two chapters' "note 1" sharing an id
# would send both references to the first.
nid = '%s-n%s-%d' % (os.path.splitext(self.cur_file or 'x')[0], num, len(self.notes))
self.notes.append((num, nid, self.convert(body)))
return ('<a class="noteref" epub:type="noteref" role="doc-noteref" '
'href="#%s" id="%sr"><sup>%s</sup></a>' % (nid, nid, num)), i
def cmd_ig(self, s, i):
a, i = balanced(s, i)
name, w, h = epub_image(a.strip())
# same ratio as print: 1px = 1/300in, text ~12pt, so 1em = 1/6in
em = h / 50.0
return ('<img class="spec" src="images/%s" alt="type specimen" '
'width="%d" height="%d" style="height:%.2fem"/>'
% (name, w, h, em)), i
def cmd_fig(self, s, i):
a, i = balanced(s, i)
name, w, h = epub_image(a.strip())
return ('<div class="figure"><img src="images/%s" alt="" width="%d" '
'height="%d"/></div>' % (name, w, h)), i
def _plate(self, num, imgpath, caption, origpage=None):
name, w, h = epub_image(imgpath.strip())
pre = ''
if origpage is not None:
anchor = 'pg-%s' % origpage
self.pages.append((origpage, self.cur_file, anchor))
pre = pagebreak(anchor, origpage)
return ('%s<figure class="plate" id="plate-%s">'
'<img src="images/%s" alt="Plate %s" width="%d" height="%d"/>'
'<figcaption>%s. %s</figcaption></figure>'
% (pre, num, name, num, w, h, num, self.convert(caption)))
def cmd_plateop(self, s, i):
pg, i = balanced(s, i)
num, i = balanced(s, i)
img, i = balanced(s, i)
cap, i = balanced(s, i)
return self._plate(num.strip(), img, cap, pg.strip()), i
def cmd_plate(self, s, i):
num, i = balanced(s, i)
img, i = balanced(s, i)
cap, i = balanced(s, i)
return self._plate(num.strip(), img, cap), i
def cmd_subhead(self, s, i):
a, i = balanced(s, i)
return '<p class="subhead"><em>' + self.convert(a) + '</em></p>', i
def cmd_subchap(self, s, i):
a, i = balanced(s, i)
return '<h3>' + self.convert(a) + '</h3>', i
def cmd_unsure(self, s, i):
a, i = balanced(s, i)
return '<span class="unsure">' + self.convert(a) + '</span>', i
def cmd_qslip(self, s, i):
a, i = balanced(s, i)
return self.convert(a), i
def cmd_erratum(self, s, i):
good, i = balanced(s, i)
_bad, i = balanced(s, i)
return self.convert(good), i
def cmd_pg(self, s, i):
a, i = balanced(s, i)
num = a.strip()
return '<a class="pgref" href="#pg-%s">%s</a>' % (num, num), i
def cmd_pl(self, s, i):
num, i = balanced(s, i)
cap, i = balanced(s, i)
pg, i = balanced(s, i)
return ('<p class="plentry"><span class="plnum">%s.</span> %s '
'<a class="pgref" href="#pg-%s">%s</a></p>'
% (num.strip(), self.convert(cap), pg.strip(), pg.strip())), i
def cmd_plx(self, s, i):
"""Like \\pl but its third argument is free markup (the List of Plates
header row), not a bare page number."""
num, i = balanced(s, i)
cap, i = balanced(s, i)
tail, i = balanced(s, i)
return ('<p class="plentry"><span class="plnum">%s.</span> %s %s</p>'
% (self.convert(num), self.convert(cap), self.convert(tail))), i
def cmd_tocl(self, s, i):
_ind, i = balanced(s, i)
label, i = balanced(s, i)
title, i = balanced(s, i)
pg, i = balanced(s, i)
return ('<p class="tocl">%s %s <a class="pgref" href="#pg-%s">%s</a></p>'
% (self.convert(label), self.convert(title), pg.strip(), pg.strip())), i
def cmd_toclnp(self, s, i):
_ind, i = balanced(s, i)
label, i = balanced(s, i)
title, i = balanced(s, i)
return ('<p class="tocl">%s %s</p>'
% (self.convert(label), self.convert(title))), i
def cmd_bibgroup(self, s, i):
a, i = balanced(s, i)
return '<h3 class="bibgroup">' + self.convert(a) + '</h3>', i
def cmd_bibhead(self, s, i):
a, i = balanced(s, i)
return '<h2 class="bibhead">' + self.convert(a) + '</h2>', i
def cmd_XeTeXglyph(self, s, i):
"""The transliteration table sets three isolated signs by raw
glyph index in Tiro Bangla, to keep the shaper from drawing a dotted
circle under them. A glyph index means nothing outside that font, so
each is mapped back to its character (read off the font's own cmap) and
set on a NO-BREAK SPACE -- Unicode's sanctioned way to show a combining
sign in isolation, which the Indic shapers accept as a base without
adding the dotted circle. Dropping them left three table cells empty.
"""
m = re.match(r'\s*(\d+)', s[i:])
if not m:
return '', i
ch = GLYPH_CHARS.get(int(m.group(1)))
if ch is None:
raise LatexError('glyph %s has no character in the font' % m.group(1), s, i)
# only a mark needs the NBSP base; a spacing letter stands on its own
base = ' ' if unicodedata.category(ch).startswith('M') else ''
return '<span class="bn">' + base + ch + '</span>', i + m.end()
# ---------------------------------------------------------------- signatures
#
# Every macro the (polished) transcription uses is declared here or has a
# cmd_ handler. convert() raises on anything else, so a new macro in the source
# is a build failure to be declared -- never text silently lost or leaked.
# Layout only: consumed with exactly (optional, mandatory) arguments, emit nothing.
LAYOUT = {
'vspace': (0, 1), 'hspace': (0, 1), 'setlength': (0, 2), 'setstretch': (0, 1),
'thispagestyle': (0, 1), 'pagestyle': (0, 1), 'pdfbookmark': (1, 2),
'begingroup': (0, 0), 'endgroup': (0, 0), 'hfill': (0, 0), 'vfill': (0, 0),
'medskip': (0, 0), 'smallskip': (0, 0), 'bigskip': (0, 0), 'noindent': (0, 0),
'par': (0, 0), 'centering': (0, 0), 'clearpage': (0, 0), 'newpage': (0, 0),
'bnsignfont': (0, 0), 'relax': (0, 0), 'protect': (0, 0),
}
# Declarations: no arguments; they restyle the rest of their group. A tag of
# None means "size or weight the reflowable text leaves to the reader".
DECLARATIONS = {
'bfseries': 'strong', 'itshape': 'em', 'slshape': 'em', 'em': 'em',
'scshape': '.smallcaps',
'mdseries': None, 'upshape': None, 'normalfont': None, 'rmfamily': None,
'small': None, 'footnotesize': None, 'scriptsize': None, 'normalsize': None,
'large': None, 'Large': None, 'LARGE': None,
}
# A \\[..] argument is consumed only if it really is a length.
LENGTH = re.compile(r'-?\d*\.?\d+\s*(em|ex|pt|pc|in|mm|cm|bp|sp|mu)')
# A run of Bengali script, spaces inside it included. The print edition sets
# ANY Bengali in Tiro Bangla (ucharclasses in src/preamble.tex); this is the
# EPUB's version of that rule, so a Bengali word is never left to whatever
# fallback font a reader has.
BENGALI_RUN = re.compile(
'[ঀ-৿](?:[ঀ-৿‌‍]|[  ](?=[ঀ-৿]))*')
class LatexError(Exception):
def __init__(self, msg, s, i):
ctx = s[max(0, i - 50):i + 50].replace('\n', ' ')
super().__init__('%s\n near: ...%s...' % (msg, ctx))
CHAPTER_MACROS = ('chapstart', 'chaphead', 'chapnum', 'partstart',
'sectionstart', 'matterstart')
def plain(tex):
"""A heading as plain text, for <title> and the navigation: markup
stripped, line breaks become a space."""
c = Converter()
h = c.convert(tex).replace('<br/>', ' ')
return html.unescape(re.sub(r'<[^>]+>', '', h)).strip()
def split_chapters(files):
"""Group page files into chapters at the heading macros."""
chapters = []
cur = {'title': 'Front matter', 'short': 'Front matter', 'kind': 'front',
'body': [], 'files': []}
for path in files:
# The polished text, not the raw transcription: tools/polish.py is the
# edition's typographic pass (range en dashes, ties, cross-reference
# links, unbreakable specimens), and the EPUB takes it from the same
# function the PDF build does, so a new rule reaches both editions.
raw = polish.polish(open(path).read())
pos = 0
while True:
m = re.search(r'\\(%s)' % '|'.join(CHAPTER_MACROS), raw[pos:])
if not m:
cur['body'].append(raw[pos:])
cur['files'].append(path)
break
start = pos + m.start()
cur['body'].append(raw[pos:start])
cur['files'].append(path)
j = start + m.end() - m.start()
cmd = m.group(1)
short = None
label = None
if cmd in ('chapstart', 'chaphead'):
short, j = opt_arg(raw, j)
title, j = balanced(raw, j)
elif cmd == 'chapnum':
a, j = balanced(raw, j)
b, j = balanced(raw, j)
# "Chapter 1" over "Charles Wilkins", as the print edition sets
# them -- joined by a line break, never by punctuation the
# original does not have (an earlier version invented ": ")
title = a + r'\\' + b
short = b
label = a
elif cmd == 'matterstart':
title, j = balanced(raw, j)
short = title
else: # partstart / sectionstart
a, j = balanced(raw, j)
b, j = balanced(raw, j)
title = a + r'\\' + b # was a + ' — ' + b: an invented dash
short = b
label = a
chapters.append(cur)
cur = {'title': title, 'short': short or title, 'kind': cmd,
'label': label, 'body': [], 'files': [path]}
pos = j
chapters.append(cur)
return [c for c in chapters if ''.join(c['body']).strip()]
ENV_RE = re.compile(r'\\begin\{(\w+)\}(.*?)\\end\{\1\}', re.S)
def extract_envs(conv, text, blocks):
"""Convert environments to HTML and stash them behind placeholders.
The generated markup must never go back through convert(), which escapes
< and >. An earlier version ran handle_envs() and then paragraphs() over
its output, so every <div>, <tr> and <blockquote> it had just produced was
re-escaped and surfaced as visible text in the reader. Placeholders keep
the two passes apart: the paragraph pass only ever sees LaTeX.
"""
def repl(m):
env, inner = m.group(1), m.group(2)
if env == 'tikzpicture':
if '\\sealbody' not in inner:
raise LatexError('a tikzpicture other than the version mark', inner, 0)
key = '@@BLOCK%d@@' % len(blocks)
blocks[key] = ('<div class="mark"><img src="images/mark.svg" '
'alt="Version mark, generated from %s" width="%d" height="%d"/></div>'
% (escape(SEAL['seed']), SEAL['w'], SEAL['h']))
return '\n\n' + key + '\n\n'
if env == 'tabular':
# the column spec contains nested braces (l@{}c@{}...), so a
# non-greedy {...} strip leaves "l@c@l@c@" glued to the first cell
k = 0
while k < len(inner) and inner[k] in ' \n\t':
k += 1
if k < len(inner) and inner[k] == '{':
_spec, k = balanced(inner, k)
inner = inner[k:]
rows = []
for line in inner.split('\\\\'):
line = line.strip()
if not line:
continue
cells = [conv.convert(c.strip()) for c in line.split('&')]
rows.append('<tr>' + ''.join('<td>%s</td>' % c for c in cells) + '</tr>')
body = '<table class="translit">' + ''.join(rows) + '</table>'
else:
inner = extract_envs(conv, inner, blocks)
if env == 'addmargin':
inner = re.sub(r'^\s*(\[[^\]]*\])?\s*(\{[^}]*\})?', '', inner)
body = paragraphs(conv, inner)
if env == 'extract':
body = '<blockquote>' + body + '</blockquote>'
elif env == 'biblist':
body = '<div class="biblist">' + body + '</div>'
elif env == 'center':
body = '<div class="center">' + body + '</div>'
elif env == 'addmargin':
body = '<div class="inset">' + body + '</div>'
key = '@@BLOCK%d@@' % len(blocks)
blocks[key] = body
return '\n\n' + key + '\n\n'
prev = None
while prev != text:
prev = text
text = ENV_RE.sub(repl, text)
return text
def restore(text, blocks):
prev = None
while prev != text:
prev = text
for k, v in blocks.items():
text = text.replace(k, v)
return text
BLOCK_START = ('<figure', '<blockquote', '<div', '<table', '<h2', '<h3', '<p class')
PAGEBREAK_ONLY = re.compile(r'(<span epub:type="pagebreak"[^>]*>[^<]*</span>\s*)+')
BLOCK_AFTER_MARKERS = re.compile(
r'(<span epub:type="pagebreak"[^>]*>[^<]*</span>\s*)+(%s)' % '|'.join(
re.escape(b) for b in ('<figure', '<blockquote', '<div', '<table',
'<h2', '<h3', '<p class', '@@BLOCK')))
def paragraphs(conv, text):
text = re.sub(r'(?m)^\s*%.*$', '', text)
out = []
pending = []
for para in re.split(r'\n\s*\n', text):
para = para.strip()
if not para:
continue
if re.fullmatch(r'@@BLOCK\d+@@', para):
# a stashed block: never re-convert it. Any page marker held back
# goes in front of it, so a break before a quotation stays before it.
out.append(''.join(pending) + para)
pending = []
continue
h = conv.convert(para).strip()
if not h:
continue
# A page break that falls between paragraphs arrives as a paragraph
# of its own. Wrapped in <p> it renders as a stray blank line (31 of
# them), so hold it and set it at the head of whatever follows.
if PAGEBREAK_ONLY.fullmatch(h):
pending.append(h)
continue
if pending:
h = ''.join(pending) + h
pending = []
if h.startswith(BLOCK_START) or h.startswith('<span epub:type="pagebreak"') \
and BLOCK_AFTER_MARKERS.match(h):
out.append(h)
else:
out.append('<p>' + h + '</p>')
out.extend(pending) # a trailing marker: keep, unwrapped
return '\n'.join(out)
XHTML = """<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE html>
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops"
lang="en" xml:lang="en">
<head><meta charset="utf-8"/><title>%s</title>
<link rel="stylesheet" type="text/css" href="style.css"/></head>
<body>
%s
</body></html>
"""
CSS = """/* The print edition's typography, carried wherever it survives reflow.
Source of each rule: src/preamble.tex. Disposition, rule by rule:
CARRIED
first-line indent 1.6em, 0.15em paragraph space -> as is
no indent after a heading or display -> as is
ragged right, no hyphenation (typescript breaks
only at spaces and typed hyphens) -> text-align:left, hyphens:none
widows and orphans forbidden -> widows/orphans 2
hanging punctuation -> hanging-punctuation (Apple
Books honours it; others ignore)
extract: indented both sides, set solid, no indent -> margins in em, tighter leading
bibliography: hanging indent 1.6em, set solid -> as is
cross-reference links: a hairline rule in linkink
(95,125,165), the text itself uncoloured -> underline, 1px, that colour
Contents / Plates whole-line links: unmarked -> as is
endnotes and plate captions a step smaller -> as is
specimens at native size (1px = 1/300in) -> height in em, per image
original page numbers in the outer margin, in sans -> floated to the right edge
of the line the page starts on
range en dashes, ties, specimen unbreakability -> from tools/polish.py itself
DROPPED, because reflow makes them meaningless or harmful
13pt XCharter on a 5.95in measure the reader sets face, size and margins
22.9pt absolute leading an absolute leading fights the reader's
own size; a relative 1.5 is kept instead
running heads there is no fixed page; readers show
their own, and the page-list carries
the original pagination for navigation
microtype expansion, raggedbottom no equivalent in CSS
*/
html { font-family: Georgia, "Times New Roman", serif; }
body { margin: 0 5%; line-height: 1.5; text-align: left;
hyphens: none; -webkit-hyphens: none; }
p { margin: 0 0 0.15em; text-indent: 1.6em; widows: 2; orphans: 2;
hanging-punctuation: first last; }
h1 + p, h2 + p, h3 + p, blockquote + p, figure + p, table + p, div + p,
.subhead + p, section > p:first-child { text-indent: 0; }
h1, h2, h3 { font-weight: bold; line-height: 1.25; margin: 1.2em 0 0.6em;
page-break-after: avoid; }
h1 { font-size: 1.5em; text-align: center; }
h2 { font-size: 1.25em; }
h3 { font-size: 1.1em; }
.subhead { font-style: italic; margin-top: 1em; text-indent: 0; }
.center { text-align: center; }
.center p { text-indent: 0; }
.inset { margin: 1em 1.5em; font-size: 0.9em; }
.inset p { text-indent: 0; }
/* extract: indented both sides (print 0.5in / 0.3in), set solid, unindented */
blockquote { margin: 0.8em 1.5em 0.8em 2.5em; line-height: 1.35; }
blockquote p { text-indent: 0; margin-bottom: 0.4em; }
/* bibliography: solid, continuation lines hang 1.6em */
.biblist p { text-indent: -1.6em; padding-left: 1.6em; margin: 0; }
h3.bibgroup { margin-top: 1em; }
h2.bibhead { text-align: center; }
/* links: a hairline rule in the print edition's linkink, text uncoloured */
a { color: inherit; }
a.xref { text-decoration: underline; text-decoration-thickness: 1px;
text-decoration-color: rgb(95,125,165); text-underline-offset: 0.18em; }
a.pgref, a.noteref { text-decoration: none; }
/* Images: width and height attributes on every <img> give the reader the real
aspect ratio; these rules let it scale without distorting -- the common EPUB
fault is a reader stretching an image to the frame. */
img { max-width: 100%; height: auto; }
figure.plate { margin: 1.2em 0; text-align: center; page-break-inside: avoid; }
figure.plate img { max-width: 100%; max-height: 88vh; width: auto; height: auto; }
figcaption { font-size: 0.85em; text-align: left; margin-top: 0.4em; }
.figure { text-align: center; margin: 1em 0; }
img.spec { vertical-align: -0.28em; width: auto; } /* height set per image */
.nowrap { white-space: nowrap; }
/* original page number: small sans at the right edge of the line the page
begins on, as the print edition sets it in the outer margin */
.pgnum { float: right; clear: right; margin: 0.25em 0 0 0.6em;
font-family: "Liberation Sans", Arial, sans-serif; font-size: 0.65em;
font-weight: normal; font-style: normal; line-height: 1;
text-indent: 0; color: #8a8a8a; } /* \\mbox: never broken */
/* Note numbers and other superscripts must not open up the line they sit on
-- the print edition's leading is even. A plain <sup> raises its baseline
and so enlarges the line box; lifting it by relative position instead
leaves the line box alone. */
sup { font-size: 0.7em; line-height: 0; vertical-align: baseline;
position: relative; top: -0.5em; }
/* Bengali in Tiro Bangla, embedded -- the print edition sets all Bengali in it */
@font-face { font-family: "Tiro Bangla"; font-style: normal; font-weight: normal;
src: url(fonts/TiroBangla-Regular.ttf); }
@font-face { font-family: "Tiro Bangla"; font-style: italic; font-weight: normal;
src: url(fonts/TiroBangla-Italic.ttf); }
.bn, :lang(bn) { font-family: "Tiro Bangla", serif; }
.bn { font-size: 1.15em; }
.bn .bn { font-size: 1em; } /* nested runs must not compound */
/* About this edition: set solid and unindented, a step smaller, as in print */
.colophon p { text-indent: 0; margin: 0 0 0.6em; font-size: 0.92em; }
.colophon .inset p, .colophon .inset { text-align: center; }
.mark { text-align: center; margin: 1em 0; }
/* the print mark stands about 2.5 text-heights tall; em keeps that proportion
at whatever size the reader sets, where its intrinsic 37px would not */
.mark img { height: 2.6em; width: auto; max-width: 100%; }
img.inline-graphic { max-height: 4em; width: auto; }
/* Errata: the original's page in a column of its own, the entry hanging */
.errata p { text-indent: 0; }
p.erratum { padding-left: 4.5em; text-indent: -4.5em; margin: 0 0 0.35em; }
p.erratum a.pgref { display: inline-block; min-width: 4.5em; text-indent: 0; }
/* Notes: one section at the back, grouped by chapter, a step smaller */
.notegroup h2 { font-size: 1.05em; margin: 1.4em 0 0.5em; }
div.note p { text-indent: 0; margin: 0 0 0.35em; font-size: 0.9em; }
a.noteback { text-decoration: none; }
.smallcaps { font-variant: small-caps; }
table.translit { border-collapse: collapse; margin: 1em 0; }
table.translit td { padding: 0.15em 0.6em 0.15em 0; vertical-align: baseline; }
.plentry, .tocl { margin: 0 0 0.3em; padding-left: 2em; text-indent: -2em; }
.plnum { display: inline-block; min-width: 2.2em; }
.unsure { border-bottom: 1px dotted #999; }
.notes { margin-top: 2em; border-top: 1px solid #ccc; padding-top: 1em;
font-size: 0.9em; }
.notes h2 { font-size: 1.1em; }
.notes p { text-indent: 0; }
aside.note { margin: 0 0 0.5em; }
"""
def finish_documents(names):
r"""Whole-book passes over the written chapters.
1. Fragment links. Anything the converter links by id -- an original page
(\pg, \xref, the Contents and List of Plates), a plate, a note -- is
emitted as a same-document "#id", because at conversion time it cannot
know which chapter file the target lands in. Resolve every one against
an index of all ids in the book; one that resolves nowhere fails the
build. (199 dead Contents and Plates links came from resolving only the
targets that happened to share a file.)
2. Empty blocks. A block holding nothing -- or only page-break markers --
renders as a stray blank line. Unwrap the markers (they stay, bare, in
the flow) and drop the empty block. Applies to every block element,
not only the <p> where it was first noticed.
"""
docs = {n: open(os.path.join(OEBPS, n)).read() for n in names}
owner = {}
for n, d in docs.items():
for i in re.findall(r'\bid="([^"]+)"', d):
owner.setdefault(i, n)
marker = r'<span epub:type="pagebreak"[^>]*>[^<]*</span>'
empty = re.compile(r'<(p|div|blockquote|section|li|figcaption)(\s[^>]*)?>'
r'((?:\s|%s)*)</\1>' % marker)
for n, d in docs.items():
def fix(m, n=n):
frag = m.group(1)
if re.search(r'\bid="%s"' % re.escape(frag), docs[n]):
return m.group(0)
if frag not in owner:
raise SystemExit('make_epub: link to #%s in %s resolves nowhere' % (frag, n))
return 'href="%s#%s"' % (owner[frag], frag)
d = re.sub(r'href="#([^"]+)"', fix, d)
prev = None
while prev != d:
prev = d
d = empty.sub(lambda m: m.group(3).strip(), d)
open(os.path.join(OEBPS, n), 'w').write(d)
ROMAN = ['I', 'II', 'III', 'IV', 'V', 'VI', 'VII', 'VIII', 'IX', 'X']
def nav_items(heads):
"""(level, label, file) for each chapter, as the PDF outline gives them.
The print edition's outline nests Parts > Sections > Chapters and labels
them by position -- I, I.A, I.A.1; II.8 where a Part has no Sections --
keeping the author's chapter numbers (src/preamble.tex, "navigation
numbering"). The same labels head the PDF's note groups ("Notes to I.A.1
Charles Wilkins"), so both the outline and the notes take them from here.
"""
items = []
part = sec = 0
for fn, ch in heads:
k = ch['kind']
name = plain(ch['short'])
if k == 'front':
continue
if k == 'partstart':
part += 1; sec = 0
items.append((1, '%s %s' % (ROMAN[part - 1], name), fn))
elif k == 'sectionstart':
sec += 1
items.append((2, '%s.%s %s' % (ROMAN[part - 1], chr(64 + sec), name), fn))
elif k == 'chapnum':
num = re.search(r'\d+', ch['label'] or '').group(0)
pos = ROMAN[part - 1] + ('.' + chr(64 + sec) if sec else '')
items.append((3 if sec else 2, '%s.%s %s' % (pos, num, name), fn))
else: # unnumbered: top level
items.append((1, name, fn))
return items
def outline(items):
return _nest(items, 0, 1)[0]
def _nest(items, i, level):
"""items[i:] as <li> elements at `level`, deeper items nested inside the
<li> that precedes them. Returns (html, next index)."""
out = []
while i < len(items) and items[i][0] >= level:
_lvl, text, fn = items[i]
out.append('<li><a href="%s">%s</a>' % (fn, escape(text)))
i += 1
if i < len(items) and items[i][0] > level:
inner, i = _nest(items, i, level + 1)
out.append('<ol>' + inner + '</ol>')
out.append('</li>')
return ''.join(out), i
def build_mark():
r"""Render the version mark to SVG, from tools/gen_seal.py itself.
The PDF draws the mark with TikZ from a hash of the sources; the EPUB takes
the same seed and the same drawing code and renders that TikZ to SVG
(XeLaTeX, then pdftocairo). Vector, so it holds at any size, and the same
mark as the PDF's whenever both are built from the same sources.
"""
import gen_seal
seed = gen_seal.content_hash()
body = gen_seal.build(seed)
tmp = os.path.join(BUILD, 'mark')
os.makedirs(tmp, exist_ok=True)
with open(os.path.join(tmp, 'mark.tex'), 'w') as f:
f.write('\\documentclass[tikz,border=2pt]{standalone}\n\\usepackage{xcolor}\n'
'\\definecolor{ink}{RGB}{26,26,28}\\definecolor{paper}{RGB}{255,255,255}\n'
'\\begin{document}\\begin{tikzpicture}[scale=0.62]\n%s\n'
'\\end{tikzpicture}\\end{document}\n' % body)
subprocess.run(['xelatex', '-interaction=nonstopmode', 'mark.tex'], cwd=tmp,
check=True, stdout=subprocess.DEVNULL)
svg = os.path.join(IMG, 'mark.svg')
subprocess.run(['pdftocairo', '-svg', os.path.join(tmp, 'mark.pdf'), svg], check=True)
m = re.search(r'<svg[^>]*\bwidth="([\d.]+)(?:pt)?"[^>]*\bheight="([\d.]+)(?:pt)?"',
open(svg).read())
w, h = (float(m.group(1)), float(m.group(2))) if m else (60.0, 60.0)
return {'seed': seed, 'w': int(round(w)), 'h': int(round(h))}
SEAL = {'seed': '', 'w': 0, 'h': 0}
def build_cover():
"""Render the PDF's cover page to a raster cover image.
The EPUB's cover is the print cover, not a re-layout of it: the mosaic's
bleed and the title's proportions only hold at the designed trim, so it is
reproduced as one image rather than reflowed.
"""
src = os.path.join(ROOT, 'ross-1988-retypeset.pdf')
if not os.path.exists(src):
return None
r = subprocess.run(['pdftoppm', '-png', '-r', '170', '-f', '1', '-l', '1', src],
capture_output=True)
if not r.stdout:
return None
import io
im = Image.open(io.BytesIO(r.stdout)).convert('RGB')
dst = os.path.join(IMG, 'cover.jpg')
im.save(dst, 'JPEG', quality=88, optimize=True)
return im.size
def convert_matter(conv, path, fname):
"""Convert one of the PDF's own front- or back-matter files (src/colophon.tex,
src/errata.tex) exactly as the pages are converted. They are read raw, as
build.sh reads them: polish.py is a pass over src/pages only."""
conv.cur_file = fname
blocks = {}
body = extract_envs(conv, open(os.path.join(ROOT, path)).read(), blocks)
return restore(paragraphs(conv, body), blocks)
def write_doc(fname, title, body):
with open(os.path.join(OEBPS, fname), 'w') as f:
f.write(XHTML % (escape(title), body))
def main():
if os.path.exists(BUILD):
shutil.rmtree(BUILD)
os.makedirs(IMG, exist_ok=True)
os.makedirs(os.path.join(OEBPS, 'fonts'), exist_ok=True)
SEAL.update(build_mark()) # the colophon's handlers read it
files = sorted(glob.glob(os.path.join(ROOT, 'src', 'pages', 'p*.tex')))
chapters = split_chapters(files)
conv = Converter()
heads, chdocs, notes_by = [], [], []
for idx, ch in enumerate(chapters):
fname = 'ch%03d.xhtml' % idx
conv.cur_file = fname
conv.notes = []
blocks = {}
body = extract_envs(conv, ''.join(ch['body']), blocks)
body = restore(paragraphs(conv, body), blocks)
# Headings go through the same converter as the text, so a two-line
# title keeps its break. The title page is its own heading, so none is
# invented for it (an earlier version printed "Front matter" there).
if ch['kind'] == 'front':
head = ''
# epub:type only: DPUB-ARIA has no title-page role, and doc-cover
# would announce this page to a screen reader as the cover image
body = '<section epub:type="titlepage">' + body + '</section>'
else:
head = '<h1>%s</h1>' % conv.convert(ch['title'])
write_doc(fname, plain(ch['short']), head + '\n' + body)
chdocs.append(fname)
heads.append((fname, ch))
if conv.notes:
notes_by.append((fname, list(conv.notes)))
items = nav_items(heads)
label = {fn: text for _lvl, text, fn in items}
# ---- About this edition: the PDF's own colophon, the EPUB's wording
conv.notes = []
write_doc('colophon.xhtml', 'About this edition',
'<section class="colophon" epub:type="colophon">%s</section>'
% convert_matter(conv, 'src/colophon.tex', 'colophon.xhtml'))
# ---- Notes: one section at the back, grouped as the PDF groups them
# ("Notes to I.A.1 Charles Wilkins"), each note linked to its reference and
# back. The PDF sets all its endnotes after the text; so does the EPUB now.
groups = []
for fn, notes in notes_by:
if fn not in label:
raise SystemExit('make_epub: notes in %s, which has no outline label' % fn)
groups.append(
'<section class="notegroup"><h2>Notes to %s</h2>%s</section>'
% (escape(label[fn]), ''.join(
'<div class="note" epub:type="endnote" id="%s"><p>'
'<a class="noteback" role="doc-backlink" href="#%sr">%s.</a> %s</p></div>'
% (nid, nid, num, txt) for num, nid, txt in notes)))
write_doc('notes.xhtml', 'Notes',
'<section epub:type="endnotes" role="doc-endnotes"><h1>Notes</h1>%s</section>'
% ''.join(groups))
# ---- Errata: src/errata.tex, the same list the PDF sets
write_doc('errata.xhtml', 'Errata',
'<section class="errata" epub:type="errata"><h1>Errata</h1>%s</section>'
% convert_matter(conv, 'src/errata.tex', 'errata.xhtml'))
order = [chdocs[0], 'colophon.xhtml'] + chdocs[1:] + ['notes.xhtml', 'errata.xhtml']
finish_documents(order)
cover_size = build_cover()
if cover_size:
cw, chh = cover_size
write_doc('cover.xhtml', 'Cover',
'<div class="center" style="margin:0;padding:0">'
'<img src="images/cover.jpg" alt="%s" width="%d" height="%d"/>'
'</div>' % (escape(TITLE), cw, chh))
# Tiro Bangla, as the PDF sets all Bengali in it. SIL OFL 1.1 (its licence
# travels in the font's own name table, which the OFL accepts); fsType 0,
# so embedding is unrestricted.
for face in ('TiroBangla-Regular.ttf', 'TiroBangla-Italic.ttf'):
shutil.copy(os.path.join(ROOT, 'fonts', face), os.path.join(OEBPS, 'fonts', face))
with open(os.path.join(OEBPS, 'style.css'), 'w') as f:
f.write(CSS)
# ---- navigation: the PDF's outline, and the page-list
toc = outline([(1, 'About this edition', 'colophon.xhtml')] + items +
[(1, 'Notes', 'notes.xhtml'), (1, 'Errata', 'errata.xhtml')])
seen = set()
plist = []
for num, fn, anchor in conv.pages:
if num in seen:
continue
seen.add(num)
plist.append('<li><a href="%s#%s">%s</a></li>' % (fn, anchor, num))
write_doc('nav.xhtml', 'Contents', """
<nav epub:type="toc" role="doc-toc" id="toc"><h1>Contents</h1><ol>%s</ol></nav>
<nav epub:type="landmarks" hidden=""><ol>
<li><a epub:type="cover" href="cover.xhtml">Cover</a></li>
<li><a epub:type="titlepage" href="%s">Title page</a></li>
<li><a epub:type="bodymatter" href="%s">Beginning</a></li>
<li><a epub:type="endnotes" href="notes.xhtml">Notes</a></li>
</ol></nav>
<nav epub:type="page-list" role="doc-pagelist" hidden=""><ol>%s</ol></nav>
""" % (toc, chdocs[0], chdocs[1], ''.join(plist)))
docs = [(fn, '') for fn in order]
# ---- package
items, spine = [], []
if cover_size:
items.append('<item id="cover" href="cover.xhtml" media-type="application/xhtml+xml"/>')
items.append('<item id="cover-image" href="images/cover.jpg" '
'media-type="image/jpeg" properties="cover-image"/>')
spine.append('<itemref idref="cover"/>')
items.append('<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" '
'properties="nav"/>')
items.append('<item id="css" href="style.css" media-type="text/css"/>')
for i, (fn, _t) in enumerate(docs):
items.append('<item id="c%d" href="%s" media-type="application/xhtml+xml"/>' % (i, fn))
spine.append('<itemref idref="c%d"/>' % i)
MT = {'.jpg': 'image/jpeg', '.png': 'image/png', '.svg': 'image/svg+xml'}
for f in sorted(os.listdir(IMG)):
if f == 'cover.jpg':
continue
items.append('<item id="img-%s" href="images/%s" media-type="%s"/>'
% (re.sub(r'\W', '_', f), f, MT[os.path.splitext(f)[1]]))
for f in sorted(os.listdir(os.path.join(OEBPS, 'fonts'))):
items.append('<item id="font-%s" href="fonts/%s" media-type="font/ttf"/>'
% (re.sub(r'\W', '_', f), f))
opf = """<?xml version="1.0" encoding="utf-8"?>
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="uid"
xml:lang="en" prefix="dcterms: http://purl.org/dc/terms/">
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
<dc:identifier id="uid">%s</dc:identifier>
<dc:title>%s</dc:title>
<dc:creator>%s</dc:creator>
<dc:language>en</dc:language>
<dc:date>2026</dc:date>
<dc:publisher>Re-typeset edition</dc:publisher>
<dc:subject>Bengali type</dc:subject>
<dc:subject>Typography</dc:subject>
<dc:subject>Printing history</dc:subject>
<dc:description>A re-typeset, searchable edition of the 1988 SOAS Ph.D. thesis, made from the ProQuest scan (number 10731406). The original pagination is preserved as EPUB page-list markers.</dc:description>
<!-- The page numbers in this book are those of the 1988 typescript; this
names the edition they come from, which is what lets a reader report
"page 214" and mean the same place another copy does. -->
<dc:source id="src">urn:proquest:10731406</dc:source>
<meta refines="#src" property="dcterms:modified">2026-01-01T00:00:00Z</meta>
<meta property="dcterms:modified">2026-01-01T00:00:00Z</meta>
<meta property="pageBreakSource" refines="#src">Ph.D. thesis, SOAS, University of London, 1988</meta>
<meta property="schema:accessMode">textual</meta>
<meta property="schema:accessMode">visual</meta>
<meta property="schema:accessibilityFeature">printPageNumbers</meta>
<meta property="schema:accessibilityFeature">tableOfContents</meta>
<meta property="schema:accessibilityFeature">longDescription</meta>
</metadata>
<manifest>%s</manifest>
<spine>%s</spine>
</package>
""" % (UID, escape(TITLE), escape(AUTHOR), ''.join(items), ''.join(spine))
with open(os.path.join(OEBPS, 'content.opf'), 'w') as f:
f.write(opf)
os.makedirs(os.path.join(BUILD, 'META-INF'), exist_ok=True)
with open(os.path.join(BUILD, 'META-INF', 'container.xml'), 'w') as f:
f.write('<?xml version="1.0" encoding="utf-8"?>\n'
'<container version="1.0" '
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n'
'<rootfiles><rootfile full-path="OEBPS/content.opf" '
'media-type="application/oebps-package+xml"/></rootfiles></container>\n')
with open(os.path.join(BUILD, 'mimetype'), 'w') as f:
f.write('application/epub+zip')
# ---- zip: mimetype first and stored, as the spec requires
if os.path.exists(OUT):
os.remove(OUT)
with zipfile.ZipFile(OUT, 'w') as z:
z.write(os.path.join(BUILD, 'mimetype'), 'mimetype', zipfile.ZIP_STORED)
for base, _dirs, fs in os.walk(BUILD):
for f in sorted(fs):
if f == 'mimetype':
continue
full = os.path.join(base, f)
z.write(full, os.path.relpath(full, BUILD), zipfile.ZIP_DEFLATED)
print('documents : %d (%d chapters + colophon, notes, errata)' % (len(docs), len(chdocs)))
print('notes : %d, in %d groups' % (sum(len(n) for _f, n in notes_by), len(notes_by)))
print('page markers : %d unique original pages' % len(plist))
print('images : %d' % len(os.listdir(IMG)))
print('epub : %s (%.1f MB)' % (os.path.relpath(OUT, ROOT),
os.path.getsize(OUT) / 1048576))
if __name__ == '__main__':
sys.exit(main())