#!/usr/bin/env python3 """ocr2tex.py PDFPAGE PRINTED [--head 'Title'] — turn tesseract scaffold into a draft page file (paragraphs joined, LaTeX specials escaped, quotes normalised). The draft MUST then be proofread against the page image and corrected.""" import sys, re, argparse ap=argparse.ArgumentParser(); ap.add_argument("page",type=int); ap.add_argument("printed",type=int) ap.add_argument("--head",default=None); ap.add_argument("--bookmark",default=None) a=ap.parse_args() txt=open(f"work/ocr/p{a.page:04d}.txt").read() lines=txt.splitlines() if a.head and lines and lines[0].strip().lower()==a.head.lower(): lines=lines[1:] body="\n".join(lines).strip() paras=[re.sub(r"\s*\n\s*"," ",p).strip() for p in re.split(r"\n\s*\n",body) if p.strip()] def esc(s): s=s.replace("\\","\\textbackslash{}") for c in "&%$#_{}": s=s.replace(c,"\\"+c) s=re.sub(r"(\d)-(\d)",r"\1-\2",s) s=s.replace("‘","`").replace("’","'").replace("“","``").replace("”","''") s=re.sub(r"\b(\w+)-\s+(\w)",r"\1\2",s) # de-hyphenate line breaks (check!) return s out=[f"% PDF page {a.page} — printed page {a.printed}",f"\\origpage{{{a.printed}}}"] if a.bookmark: out.append(f"\\pdfbookmark[1]{{{a.bookmark}}}{{bm{a.page}}}") if a.head: out.append(f"\\chaphead{{{a.head}}}") out.append("\n\n".join(esc(p) for p in paras)) p=f"src/pages/p{a.page:04d}.tex"; open(p,"w").write("\n".join(out)+"\n"); print(p, len(paras),"paras")