A searchable, re-typeset edition of Fiona Ross, "The Evolution of the Printed Bengali Character from 1778 to 1978" (Ph.D., SOAS, 1988), transcribed from the 431-leaf ProQuest scan. All 431 pages done; 178 plates and 410 inline type specimens cut from the scan; 51 errata. Tracked: the transcription (src/pages), the preamble and its typographic decisions, the cut images (plates/ — not reliably regenerable, the crop specs for the inline cuts were never scripted), tools, and the four working documents. Not tracked: the built PDF, which `make` remakes from src/ and plates/; the ProQuest scan under source/, which is third-party and needed only by `make prep` and `make plate`; scans/ and work/, both regenerable. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
26 lines
1.4 KiB
Python
Executable File
26 lines
1.4 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
"""ocr2tex.py PDFPAGE PRINTED [--head 'Title'] — turn tesseract scaffold into a draft
|
||
page file (paragraphs joined, LaTeX specials escaped, quotes normalised).
|
||
The draft MUST then be proofread against the page image and corrected."""
|
||
import sys, re, argparse
|
||
ap=argparse.ArgumentParser(); ap.add_argument("page",type=int); ap.add_argument("printed",type=int)
|
||
ap.add_argument("--head",default=None); ap.add_argument("--bookmark",default=None)
|
||
a=ap.parse_args()
|
||
txt=open(f"work/ocr/p{a.page:04d}.txt").read()
|
||
lines=txt.splitlines()
|
||
if a.head and lines and lines[0].strip().lower()==a.head.lower(): lines=lines[1:]
|
||
body="\n".join(lines).strip()
|
||
paras=[re.sub(r"\s*\n\s*"," ",p).strip() for p in re.split(r"\n\s*\n",body) if p.strip()]
|
||
def esc(s):
|
||
s=s.replace("\\","\\textbackslash{}")
|
||
for c in "&%$#_{}": s=s.replace(c,"\\"+c)
|
||
s=re.sub(r"(\d)-(\d)",r"\1-\2",s)
|
||
s=s.replace("‘","`").replace("’","'").replace("“","``").replace("”","''")
|
||
s=re.sub(r"\b(\w+)-\s+(\w)",r"\1\2",s) # de-hyphenate line breaks (check!)
|
||
return s
|
||
out=[f"% PDF page {a.page} — printed page {a.printed}",f"\\origpage{{{a.printed}}}"]
|
||
if a.bookmark: out.append(f"\\pdfbookmark[1]{{{a.bookmark}}}{{bm{a.page}}}")
|
||
if a.head: out.append(f"\\chaphead{{{a.head}}}")
|
||
out.append("\n\n".join(esc(p) for p in paras))
|
||
p=f"src/pages/p{a.page:04d}.tex"; open(p,"w").write("\n".join(out)+"\n"); print(p, len(paras),"paras")
|