A searchable, re-typeset edition of Fiona Ross, "The Evolution of the Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988), transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178 plates and 410 inline type specimens cut from the scan; 51 errata. Tracked: the transcription (src/pages), the preamble and its typographic decisions, the cut images (plates/ — not reliably regenerable, the crop specs for the inline cuts were never scripted), tools, and the four working documents. Not tracked: the built PDF, which `make` remakes from src/ and plates/; the ProQuest scan under source/, which is third-party and needed only by `make prep` and `make plate`; scans/ and work/, both regenerable. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
55 lines
3.3 KiB
Python
Executable File
55 lines
3.3 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
r"""Ratchet checks for the re-typeset project. Exit 1 on any hard failure.
|
|
- every manifest row marked done points at an existing page file (or is dropped/colophon)
|
|
- every src/pages file is a manifest row marked done
|
|
- page file carries \origpage{P}/\plateop{P} with P == manifest.printed
|
|
- a head macro (\chaphead/\chapstart/\chapnum/\partstart/\sectionstart) precedes \origpage
|
|
- printed page numbers increase in PDF order; note numbers \fn{n} run 1,2,3… per chapter
|
|
- lists \unsure{} readings so they can be mirrored in QUESTIONS.md
|
|
- every \erratum{corrected}{as printed} has a matching \erratumline in src/errata.tex
|
|
"""
|
|
import csv, re, glob, os, sys
|
|
fail=[]; warn=[]
|
|
rows=list(csv.DictReader(open("manifest.tsv"),delimiter="\t"))
|
|
files=set(glob.glob("src/pages/p*.tex"))
|
|
seen=set(); last_printed=0; fn_expect=1; unsure=[]; errata_used=[]
|
|
for r in sorted(rows,key=lambda r:int(r["pdf_page"])):
|
|
p=int(r["pdf_page"]); st=r["status"]; kind=r["kind"]; f=r["src_file"]
|
|
if st!="done": continue
|
|
if kind in ("dropped","colophon"): continue
|
|
if not f or not os.path.exists(f): fail.append(f"PDF {p}: done but src_file missing ({f!r})"); continue
|
|
seen.add(f); s=open(f).read()
|
|
m=re.search(r"\\(?:origpage|plateop)\{(\d+)\}",s)
|
|
if not m: fail.append(f"{f}: no \\origpage/\\plateop")
|
|
else:
|
|
pr=int(m.group(1))
|
|
if r["printed"] and pr!=int(r["printed"]): fail.append(f"{f}: \\origpage{{{pr}}} but manifest printed={r['printed']}")
|
|
if pr<=last_printed: fail.append(f"{f}: printed {pr} not after previous {last_printed}")
|
|
last_printed=pr
|
|
heads = ("\\chaphead", "\\chapstart", "\\partstart", "\\sectionstart", "\\chapnum")
|
|
if any(h in s for h in heads):
|
|
if s.find("\\origpage")!=-1 and s.find("\\origpage")<max(s.find(h) for h in heads):
|
|
fail.append(f"{f}: \\origpage appears before \\chaphead/\\chapstart (would leave an empty page)")
|
|
fn_expect=1
|
|
for n in re.findall(r"\\fn\{(\d+)\}",s):
|
|
if int(n)!=fn_expect: fail.append(f"{f}: \\fn{{{n}}} but expected {fn_expect}")
|
|
fn_expect=int(n)+1
|
|
unsure+= [f"{f}: {u}" for u in re.findall(r"\\unsure\{([^}]*)\}",s)]
|
|
# \erratum args must be plain text (no nested braces): put \emph outside it
|
|
errata_used += [(r["printed"], corr, orig, f)
|
|
for corr, orig in re.findall(r"\\erratum\{([^}]*)\}\{([^}]*)\}", s)]
|
|
errata_src = open("src/errata.tex").read() if os.path.exists("src/errata.tex") else ""
|
|
listed = set(re.findall(r"\\erratumline\{(\d+)\}\{([^}]*)\}\{([^}]*)\}", errata_src))
|
|
for pg, corr, orig, f in errata_used:
|
|
if (pg, orig, corr) not in listed:
|
|
fail.append(f"{f}: \\erratum{{{corr}}}{{{orig}}} has no \\erratumline{{{pg}}}{{{orig}}}{{{corr}}} in src/errata.tex")
|
|
for pg, orig, corr in sorted(listed):
|
|
if not any(pg==u[0] and orig==u[2] and corr==u[1] for u in errata_used):
|
|
fail.append(f"src/errata.tex: \\erratumline{{{pg}}}{{{orig}}}{{{corr}}} has no \\erratum in a page file")
|
|
for f in sorted(files-seen): fail.append(f"{f} exists but manifest row is not done/linked")
|
|
done=sum(1 for r in rows if r["status"]=="done"); print(f"progress: {done}/{len(rows)} PDF pages done")
|
|
if unsure: print("unsure readings (mirror in QUESTIONS.md):"); [print(" ",u) for u in unsure]
|
|
for w in warn: print("WARN",w)
|
|
for e in fail: print("FAIL",e)
|
|
sys.exit(1 if fail else 0)
|