#!/usr/bin/env python3 # PDFVerge PDF to Word converter # Copyright (C) 2026 WebScare (Pvt.) Limited # # This program is free software: you can redistribute it and/or modify it # under the terms of the GNU Affero General Public License as published by the # Free Software Foundation, either version 3 of the License, or (at your # option) any later version. It is distributed WITHOUT ANY WARRANTY; see # . Source: https://pdfverge.com/source/ """ PDFVerge — PDF to editable Word (DOCX) converter. pdf2word.py INPUT.pdf OUTPUT.docx [--ocr auto|on|off] [--lang eng] [--max-pages 200] [--mem-mb 2048] [--cpu-secs 170] Called by the api.pdfverge.com route as a separate process, one file per call. Result contract exit 0 stdout = {"ok": true, "pages": N, "mode": "layout"|"ocr", "scanned_pages": K, "lists_fixed": L, "seconds": S} exit 2 stdout = {"ok": false, "error": ""} exit 3 stdout = {"ok": false, "error": "...", "internal": ""} Engines layout pdf2docx rebuilds real paragraphs, tables, images and columns. This is the normal path for PDFs made from Word, Google Docs, etc. ocr for scanned PDFs (pages that are only a picture). Tesseract reads the text as editable paragraphs; logos, signatures, stamps and photos are cropped from the scan and kept in place beside it. Licence note: pdf2docx is MIT, but it runs on PyMuPDF, which is AGPL-3.0. Because PDFVerge offers this over the network, publish this file's source (it is self-contained) or buy a PyMuPDF commercial licence from Artifex. """ import argparse import json import logging import os import sys import time # ----------------------------------------------------------------- helpers BULLET_GLYPHS = set("•●▪■◦‣⁃∙·○–-*" "") PUA_BULLETS = {"": "•", "": "▪", "": "•", "": "•", "": "✓", "": "▪", "": "▪", "": "•"} # stdout carries exactly one JSON line. Libraries (PyMuPDF, pdf2docx, MuPDF's # C code) print warnings to fd 1, so the real stdout is set aside and fd 1 # is pointed at /dev/null for the whole run. _RESULT_FD = os.dup(1) _devnull = os.open(os.devnull, os.O_WRONLY) os.dup2(_devnull, 1) def say(obj, code): os.write(_RESULT_FD, (json.dumps(obj) + "\n").encode()) sys.stdout.flush() os._exit(code) def fail(msg, code=2, internal=None): out = {"ok": False, "error": msg} if internal: out["internal"] = internal[:2000] say(out, code) def limit_resources(mem_mb, cpu_secs): """A malformed PDF must not take the whole server down with it.""" try: import resource if mem_mb: b = mem_mb * 1024 * 1024 resource.setrlimit(resource.RLIMIT_AS, (b, b)) if cpu_secs: resource.setrlimit(resource.RLIMIT_CPU, (cpu_secs, cpu_secs + 5)) except Exception: pass # not Linux; the caller's timeout still applies def is_bullet_marker(text): t = "".join(text.split()) if not t: return False if all(ch in BULLET_GLYPHS for ch in t): return True # "1.2.3." or "a)b)c)" style numbering stacked in one cell import re return re.fullmatch(r"(?:\(?[0-9ivxlcdmIVXLCDM]{1,4}[.)]|\(?[a-zA-Z][.)])+", t) is not None def page_is_scan(page): """True when a page carries (almost) no real text but is covered by a picture.""" text = page.get_text("text").strip() if len(text) > 40: return False area = abs(page.rect) if area <= 0: return False covered = 0.0 for info in page.get_image_info(): r = info.get("bbox") if r: x0, y0, x1, y1 = r covered += max(0.0, x1 - x0) * max(0.0, y1 - y0) return covered / area > 0.5 # ------------------------------------------------------------ post-process def _split_on_breaks(p_el): """One w:p whose items are separated by soft line breaks -> list of w:p, one per item, keeping each run's formatting.""" from copy import deepcopy from docx.oxml.ns import qn ppr = p_el.find(qn("w:pPr")) out, cur = [], None def new_p(): p = deepcopy(p_el) for ch in list(p): if ch.tag != qn("w:pPr"): p.remove(ch) return p cur = new_p() for r in p_el.iterchildren(): if r.tag == qn("w:pPr"): continue if r.tag != qn("w:r"): cur.append(deepcopy(r)) # hyperlinks, bookmarks: keep whole continue rpr = r.find(qn("w:rPr")) buf = deepcopy(r) for ch in list(buf): if ch.tag != qn("w:rPr"): buf.remove(ch) for ch in r.iterchildren(): if ch.tag == qn("w:rPr"): continue if ch.tag == qn("w:br") and ch.get(qn("w:type")) in (None, "textWrapping"): if len(buf) > (1 if rpr is not None else 0): cur.append(buf) out.append(cur) cur = new_p() buf = deepcopy(r) for c2 in list(buf): if c2.tag != qn("w:rPr"): buf.remove(c2) continue buf.append(deepcopy(ch)) if len(buf) > (1 if rpr is not None else 0): cur.append(buf) out.append(cur) return out def _text_of(p_el): from docx.oxml.ns import qn return "".join(t.text or "" for t in p_el.iter(qn("w:t"))) def tidy_docx(path): """Fix the two things pdf2docx most often gets wrong for Word users: 1. a bullet or numbered list detected as a borderless table (marker column(s) | text column) -> real list paragraphs again 2. bullets drawn with private-use Symbol/Wingdings code points, which show as empty boxes as soon as the font changes.""" import re from copy import deepcopy from docx import Document from docx.oxml.ns import qn from docx.shared import Pt from docx.text.paragraph import Paragraph doc = Document(path) fixed = 0 for tbl in list(doc.tables): ncols = len(tbl.columns) if ncols < 2 or not tbl.rows: continue seen, marker_cells, text_cells, ok = set(), [], [], True for row in tbl.rows: cells = row.cells for ci, cell in enumerate(cells): if id(cell._tc) in seen: continue seen.add(id(cell._tc)) if ci < len(cells) - 1: if cell.text.strip() and not is_bullet_marker(cell.text): ok = False marker_cells.append(cell) else: text_cells.append(cell) if not ok or not text_cells: continue markers = [] for c in marker_cells: markers += [m for m in re.split(r"\s+", c.text) if m] if not markers: continue items = [] for c in text_cells: for p in c.paragraphs: items += [x for x in _split_on_breaks(p._p) if _text_of(x).strip()] if not items: continue per_item = len(markers) == len(items) anchor = tbl._tbl for i, item in enumerate(items): m = markers[i] if per_item else markers[0] m = PUA_BULLETS.get(m, m) anchor.addprevious(item) para = Paragraph(item, doc._body) first = para.runs[0] if para.runs else None mark = para.add_run(m + "\t") if first is not None: first._r.addprevious(mark._r) rpr = first._r.find(qn("w:rPr")) if rpr is not None: mark._r.insert(0, deepcopy(rpr)) mark.bold = False mark.italic = False last = para.runs[-1] if para.runs else None if last is not None and last.text.endswith(" "): last.text = last.text.rstrip() pf = para.paragraph_format pf.left_indent = Pt(18) pf.first_line_indent = Pt(-18) pf.right_indent = Pt(0) # the table cell's right edge no longer applies anchor.getparent().remove(anchor) fixed += 1 # swap private-use bullet glyphs for real Unicode ones everywhere def fix_runs(paragraphs): for p in paragraphs: for run in p.runs: if any(ch in run.text for ch in PUA_BULLETS): t = run.text for k, v in PUA_BULLETS.items(): t = t.replace(k, v) run.text = t if run.font.name in ("Symbol", "Wingdings"): run.font.name = None fix_runs(doc.paragraphs) for tbl in doc.tables: for row in tbl.rows: for cell in row.cells: fix_runs(cell.paragraphs) doc.save(path) return fixed # ------------------------------------------------------------------- OCR OCR_BULLET_TOKENS = {"e", "¢", "«", "»", "°", "*", "o", "©", "@", "•", "·", "-", "–", "■", "▪", "e¢"} OCR_DPI = 300 def _tesseract_tsv(png_path, lang): import subprocess env = dict(os.environ, OMP_THREAD_LIMIT="1") # one core per job on a shared box r = subprocess.run(["tesseract", png_path, "stdout", "-l", lang, "--psm", "3", "tsv"], capture_output=True, text=True, timeout=120, env=env) if r.returncode != 0: raise RuntimeError("tesseract: " + (r.stderr or "")[-500:]) rows = [] lines = r.stdout.splitlines() head = lines[0].split("\t") if lines else [] for ln in lines[1:]: f = ln.split("\t") if len(f) != len(head): continue d = dict(zip(head, f)) if d.get("level") != "5": continue txt = d.get("text", "").strip() conf = float(d.get("conf", "-1")) if not txt or conf < 0: continue rows.append({"b": int(d["block_num"]), "p": int(d["par_num"]), "l": int(d["line_num"]), "x": int(d["left"]), "y": int(d["top"]), "w": int(d["width"]), "h": int(d["height"]), "c": conf, "t": txt}) return rows def _ocr_paragraphs(words): """Tesseract words -> list of items (kind, text, bbox). kind: 'p' paragraph, 'row' table-like line (large gaps become tabs), 'bullet' list item. bbox = (x0, y0, x1, y1) in image pixels, used to place graphics beside text.""" from collections import OrderedDict pars = OrderedDict() for w in words: pars.setdefault((w["b"], w["p"]), OrderedDict()).setdefault(w["l"], []).append(w) out = [] for lines in pars.values(): text_lines, line_boxes, line_h, is_table = [], [], [], False for ws in lines.values(): ws.sort(key=lambda w: w["x"]) h = sorted(w["h"] for w in ws)[len(ws) // 2] or 10 parts, gap_found = [ws[0]["t"]], False for a, b in zip(ws, ws[1:]): gap = b["x"] - (a["x"] + a["w"]) if gap > 2.2 * h: parts.append("\t") gap_found = True else: parts.append(" ") parts.append(b["t"]) text_lines.append("".join(parts)) line_h.append(h) line_boxes.append((min(w["x"] for w in ws), min(w["y"] for w in ws), max(w["x"] + w["w"] for w in ws), max(w["y"] + w["h"] for w in ws))) is_table = is_table or gap_found if is_table: for tl, lb, lh in zip(text_lines, line_boxes, line_h): k, t = _classify(tl, "row" if "\t" in tl else "p") out.append((k, t, lb, lh)) continue chunks = [] # [text, bbox, word height] prev = None for tl, lb, lhh in zip(text_lines, line_boxes, line_h): lh = max(1, lb[3] - lb[1]) gap = (lb[1] - prev[3]) if prev else 0 if not chunks or _starts_item(tl) or gap > 0.9 * lh: chunks.append([tl, list(lb), lhh]) else: if chunks[-1][0].endswith("-") and tl[:1].islower(): chunks[-1][0] = chunks[-1][0][:-1] + tl else: chunks[-1][0] += " " + tl bb = chunks[-1][1] chunks[-1][1] = [min(bb[0], lb[0]), min(bb[1], lb[1]), max(bb[2], lb[2]), max(bb[3], lb[3])] prev = lb for c, bb, ch in chunks: k, t = _classify(c, "p") out.append((k, t, tuple(bb), ch)) return out _NUM_ITEM = None def _starts_item(line): import re global _NUM_ITEM if _NUM_ITEM is None: _NUM_ITEM = re.compile(r"^(?:\(?\d{1,3}[.)]|\(?[a-z][.)]|[ivx]{1,4}[.)])\s") first = line.replace("\t", " ").split(" ", 1)[0] return bool(_NUM_ITEM.match(line.replace("\t", " "))) or first in OCR_BULLET_TOKENS and len(line) > 2 \ and line.replace("\t", " ").split(" ", 1)[-1][:1].isupper() def _classify(text, kind): first, _, rest = text.replace("\t", " ", 1).partition(" ") rest = rest.lstrip() if first in OCR_BULLET_TOKENS and rest[:1].isupper(): return ("bullet", rest) return (kind, text) def _split_words(words, rgb=None): """Separate real text from 'words' Tesseract invented while reading pictures: low-confidence junk from icons, signatures and stamps, and big coloured logo lettering. Big, dark, confidently read words are headings and stay text. Returns (text_words, graphic_words, median_word_height).""" good = sorted(w["h"] for w in words if w["c"] >= 60 and sum(ch.isalnum() for ch in w["t"]) >= 2) med = good[len(good) // 2] if good else 30 text, graphic = [], [] for w in words: alnum = sum(ch.isalnum() for ch in w["t"]) junk = w["c"] < 40 or (alnum < 2 and w["c"] < 80 and len(w["t"]) > 1) huge = w["h"] > 2.4 * med if huge and not junk and w["c"] >= 85 and alnum >= 3 and rgb is not None and not _is_coloured(rgb, w): huge = False (graphic if (junk or huge) else text).append(w) return text, graphic, med def _is_coloured(rgb, w): """True when the dark pixels of a word are tinted (logo lettering), not black/grey.""" import numpy as np reg = rgb[w["y"]:w["y"] + w["h"], w["x"]:w["x"] + w["w"]].astype(np.int16) if reg.size == 0: return False mx, mn = reg.max(axis=2), reg.min(axis=2) inkpx = mx < 200 if inkpx.sum() < 20: inkpx = (mx - mn) > 45 if inkpx.sum() < 20: return False return float(((mx - mn)[inkpx] > 45).mean()) > 0.5 def _cv2(): import cv2 cv2.setNumThreads(1) # a big server has many cores; one converter job must not start a thread per core return cv2 def _find_graphics(rgb, text_words): """Boxes of non-text ink on a scanned page (logos, signatures, stamps, photos, coloured bars). Text-word boxes are erased first; what remains is grouped.""" import numpy as np cv2 = _cv2() H, W = rgb.shape[:2] gray = cv2.cvtColor(rgb, cv2.COLOR_RGB2GRAY) mx = rgb.max(axis=2).astype(np.int16) mn = rgb.min(axis=2).astype(np.int16) colored = ((mx - mn) > 45) & (mx < 250) # tinted ink: logos, stamps, blue signatures ink = ((gray < 165) | colored).astype(np.uint8) for w in text_words: x0, y0 = max(0, w["x"] - 4), max(0, w["y"] - 4) ink[y0:w["y"] + w["h"] + 4, x0:w["x"] + w["w"] + 4] = 0 # tinted panels (a chart on a grey card, a coloured bar) count as one picture light = ((gray < 242) | colored).astype(np.uint8) lk = cv2.getStructuringElement(cv2.MORPH_RECT, (9, 9)) light = cv2.morphologyEx(light, cv2.MORPH_OPEN, lk) ln, _, lstats, _ = cv2.connectedComponentsWithStats(light, 8) panels = [] for i in range(1, ln): x, y, w, h, area = lstats[i] if w * h >= 0.015 * W * H and w * h <= 0.6 * W * H and area >= 0.6 * w * h and min(w, h) >= 60: panels.append((x, y, w, h)) k = cv2.getStructuringElement(cv2.MORPH_RECT, (31, 31)) closed = cv2.morphologyEx(ink, cv2.MORPH_CLOSE, k) for x, y, w, h in panels: closed[y:y + h, x:x + w] = 1 n, _, stats, _ = cv2.connectedComponentsWithStats(closed, 8) boxes = [] for i in range(1, n): x, y, w, h, _ = stats[i] if w * h > 0.6 * W * H: continue if max(w, h) < 40 or w * h < 1200 or min(w, h) < 15: continue # specks, underlines, dust if (x <= 3 or y <= 3 or x + w >= W - 3 or y + h >= H - 3) and min(w, h) < 60: continue # scanner edge shadows density = ink[y:y + h, x:x + w].mean() is_panel = any(px <= x + 2 and py <= y + 2 and px + pw >= x + w - 2 and py + ph >= y + h - 2 for px, py, pw, ph in panels) if density < 0.015 and not is_panel: continue # a ruled table or a frame around text: thin lines with many words inside stays editable text inside = sum(1 for t in text_words if x <= t["x"] + t["w"] / 2 <= x + w and y <= t["y"] + t["h"] / 2 <= y + h) if inside >= 6 and density < 0.12 and not is_panel: continue boxes.append([x, y, x + w, y + h]) # merge boxes that overlap or nearly touch (a logo mark and its lettering) merged = True while merged: merged = False for i in range(len(boxes)): for j in range(i + 1, len(boxes)): a, b = boxes[i], boxes[j] near = a[0] - 20 < b[2] and b[0] - 20 < a[2] and a[1] - 20 < b[3] and b[1] - 20 < a[3] ox = min(a[2], b[2]) - max(a[0], b[0]) wa, wb = a[2] - a[0], b[2] - b[0] stacked = (ox > 0.5 * min(wa, wb) and min(wa, wb) > 0.5 * max(wa, wb) # a column of icons and a[1] - 70 < b[3] and b[1] - 70 < a[3]) if near or stacked: boxes[i] = [min(a[0], b[0]), min(a[1], b[1]), max(a[2], b[2]), max(a[3], b[3])] boxes.pop(j) merged = True break if merged: break pad = 8 return [(max(0, x0 - pad), max(0, y0 - pad), min(W, x1 + pad), min(H, y1 + pad)) for x0, y0, x1, y1 in boxes] def _crop_png(rgb, box, path, text_words=()): cv2 = _cv2() x0, y0, x1, y1 = box crop = rgb[y0:y1, x0:x1].copy() for w in text_words: # text that stays editable is blanked out of the picture, so it isn't doubled ax0, ay0 = max(x0, w["x"] - 2), max(y0, w["y"] - 2) ax1, ay1 = min(x1, w["x"] + w["w"] + 2), min(y1, w["y"] + w["h"] + 2) if ax0 < ax1 and ay0 < ay1: crop[ay0 - y0:ay1 - y0, ax0 - x0:ax1 - x0] = 255 scale = 200.0 / OCR_DPI # 200 dpi is plenty in Word, keeps files small crop = cv2.resize(crop, (max(1, int(crop.shape[1] * scale)), max(1, int(crop.shape[0] * scale))), interpolation=cv2.INTER_AREA) cv2.imwrite(path, cv2.cvtColor(crop, cv2.COLOR_RGB2BGR), [cv2.IMWRITE_PNG_COMPRESSION, 9]) def _write_items(container, items, add_paragraph, med=None, base_pt=11): from docx.shared import Pt for kind, text, _, h in items: para = add_paragraph(("•\t" + text) if kind == "bullet" else text) pf = para.paragraph_format if kind == "bullet": pf.left_indent, pf.first_line_indent = Pt(18), Pt(-18) if kind == "row": pf.space_after = Pt(0) for n in range(1, 8): pf.tab_stops.add_tab_stop(Pt(110 * n)) if med and kind == "p" and h > 1.4 * med and len(text.split()) <= 14: # headings keep their relative size size = Pt(min(30, round(base_pt * h / med))) for r in para.runs: r.font.size = size r.font.bold = True def ocr_to_docx(src, out_path, lang): """Scanned pages: editable text from Tesseract, plus the page's pictures (logo, signature, stamp, photos) cropped from the scan and placed where they were: side by side with text in a borderless table when they share a band of the page (a letterhead), otherwise in reading order.""" import shutil import tempfile import numpy as np import pymupdf from docx import Document from docx.enum.table import WD_TABLE_ALIGNMENT from docx.enum.text import WD_BREAK, WD_ALIGN_PARAGRAPH from docx.oxml import OxmlElement from docx.oxml.ns import qn from docx.shared import Pt if not shutil.which("tesseract"): raise RuntimeError("tessdata: tesseract binary not installed") doc = Document() normal = doc.styles["Normal"] normal.font.name = "Calibri" normal.font.size = Pt(11) sec = doc.sections[0] r0 = src[0].rect sec.page_width, sec.page_height = Pt(r0.width), Pt(r0.height) margin = 36 sec.left_margin = sec.right_margin = Pt(margin) sec.top_margin = sec.bottom_margin = Pt(margin) px2pt = 72.0 / OCR_DPI last = len(src) - 1 base_pt, page_med = 11, None with tempfile.TemporaryDirectory(prefix="pv-ocr-") as tmp: for i, page in enumerate(src): pix = page.get_pixmap(dpi=OCR_DPI, colorspace=pymupdf.csRGB, alpha=False) rgb = np.frombuffer(pix.samples, dtype=np.uint8).reshape(pix.height, pix.width, 3).copy() png = os.path.join(tmp, "p%d.png" % i) pymupdf.Pixmap(pymupdf.csGRAY, pix).save(png) del pix words = _tesseract_tsv(png, lang) os.remove(png) text_words, _, med = _split_words(words, rgb) if i == 0: # body font size and spacing from the scan, so a one-page letter stays one page base_pt = max(9, min(12, round(med * px2pt * 1.35))) normal.font.size = Pt(base_pt) page_med = med normal.paragraph_format.space_after = Pt(4) normal.paragraph_format.line_spacing = 1.0 boxes = _find_graphics(rgb, text_words) # lettering that belongs to a logo (large text touching a graphic) joins the graphic for _ in range(2): grow = [w for w in text_words if w["h"] > 1.6 * med and any( w["x"] - 25 < b[2] and b[0] - 25 < w["x"] + w["w"] and w["y"] - 25 < b[3] and b[1] - 25 < w["y"] + w["h"] for b in boxes)] if not grow: break text_words = [w for w in text_words if w not in grow] boxes = _find_graphics(rgb, text_words) def inside(w): cx, cy = w["x"] + w["w"] / 2, w["y"] + w["h"] / 2 return any(b[0] <= cx <= b[2] and b[1] <= cy <= b[3] for b in boxes) text_words = [w for w in text_words if not inside(w)] # text printed inside a logo stays in the picture elements = [] # (y0, y1, x0, x1, kind, payload, seq) for seq, it in enumerate(_ocr_paragraphs(text_words)): x0, y0, x1, y1 = it[2] elements.append((y0, y1, x0, x1, "text", it, seq)) for j, b in enumerate(boxes): path = os.path.join(tmp, "g%d_%d.png" % (i, j)) _crop_png(rgb, b, path, text_words) elements.append((b[1], b[3], b[0], b[2], "img", path, 10000 + j)) del rgb elements.sort(key=lambda e: (e[0], e[6])) page_w_px = page.rect.width / px2pt content_pt = page.rect.width - 2 * margin # group into horizontal bands of the page bands = [] for e in elements: ov = (bands[-1]["y1"] - e[0]) if bands else 0 if bands and ov > 6 and ov > 0.3 * (e[1] - e[0]): bands[-1]["els"].append(e) bands[-1]["y1"] = max(bands[-1]["y1"], e[1]) else: bands.append({"y1": e[1], "els": [e]}) def add_img(paragraph, e, max_pt): w_pt = min((e[3] - e[2]) * px2pt, max_pt) paragraph.add_run().add_picture(e[5], width=Pt(w_pt)) for band in bands: els = band["els"] has_img = any(e[4] == "img" for e in els) if not has_img: _write_items(doc, [e[5] for e in sorted(els, key=lambda e: e[6])], doc.add_paragraph, page_med, base_pt) continue # columns: elements whose horizontal extents overlap share a column cols = [] for e in sorted(els, key=lambda e: e[2]): if cols and e[2] < cols[-1]["x1"] + 30: cols[-1]["els"].append(e) cols[-1]["x1"] = max(cols[-1]["x1"], e[3]) else: cols.append({"x0": e[2], "x1": e[3], "els": [e]}) if len(cols) == 1: for e in sorted(els, key=lambda e: (e[0], e[6])): if e[4] == "text": _write_items(doc, [e[5]], doc.add_paragraph, page_med, base_pt) else: para = doc.add_paragraph() cx = (e[2] + e[3]) / 2 / page_w_px para.alignment = WD_ALIGN_PARAGRAPH.LEFT if cx < 0.36 else (WD_ALIGN_PARAGRAPH.RIGHT if cx > 0.64 else WD_ALIGN_PARAGRAPH.CENTER) add_img(para, e, content_pt) continue # side by side: a borderless table, one cell per column spans = [] for k, c in enumerate(cols): right = cols[k + 1]["x0"] if k + 1 < len(cols) else page_w_px left = c["x0"] if k else 0 spans.append(max(1.0, right - left)) total = sum(spans) widths = [content_pt * s / total for s in spans] table = doc.add_table(rows=1, cols=len(cols)) table.alignment = WD_TABLE_ALIGNMENT.CENTER table.autofit = False tblPr = table._tbl.tblPr layout = OxmlElement("w:tblLayout") layout.set(qn("w:type"), "fixed") tblPr.append(layout) for k, c in enumerate(cols): table.columns[k].width = Pt(widths[k]) # the grid decides widths in a fixed layout cell = table.rows[0].cells[k] cell.width = Pt(widths[k]) first = True for e in sorted(c["els"], key=lambda e: (e[0], e[6])): if e[4] == "text": def cell_par(t, cell=cell): nonlocal first if first: first = False p = cell.paragraphs[0] p.add_run(t) return p return cell.add_paragraph(t) _write_items(cell, [e[5]], cell_par, page_med, base_pt) else: p = cell.paragraphs[0] if first else cell.add_paragraph() first = False add_img(p, e, widths[k] - 8) doc.add_paragraph().paragraph_format.space_after = Pt(0) if not elements: doc.add_paragraph("") if i < last: doc.add_paragraph().add_run().add_break(WD_BREAK.PAGE) doc.save(out_path) # ------------------------------------------------------------------- main def main(): ap = argparse.ArgumentParser() ap.add_argument("input") ap.add_argument("output") ap.add_argument("--ocr", choices=["auto", "on", "off"], default="auto") ap.add_argument("--lang", default="eng", help="Tesseract language(s), e.g. eng or eng+ara") ap.add_argument("--max-pages", type=int, default=200) ap.add_argument("--max-ocr-pages", type=int, default=20, help="scanned PDFs take ~2 s a page to read; keep this under the API timeout") ap.add_argument("--mem-mb", type=int, default=2048) ap.add_argument("--cpu-secs", type=int, default=170) a = ap.parse_args() limit_resources(a.mem_mb, a.cpu_secs) logging.disable(logging.CRITICAL) # pdf2docx logs every page to stderr t0 = time.time() try: with open(a.input, "rb") as fh: head = fh.read(1024) except OSError as e: fail("The uploaded file could not be read.", 3, repr(e)) if b"%PDF" not in head: fail("That file is not a PDF. Choose a .pdf file and try again.") import warnings warnings.filterwarnings("ignore") import pymupdf try: src = pymupdf.open(a.input) except Exception as e: fail("This PDF is damaged and could not be opened.", 2, repr(e)) if src.needs_pass: fail("This PDF is protected with an open password. Remove the password first, then convert it.") n = len(src) if n == 0: fail("This PDF has no pages.") if n > a.max_pages: fail("This PDF has %d pages. The limit is %d pages per file — split it first, then convert each part." % (n, a.max_pages)) scanned = sum(1 for p in src if page_is_scan(p)) use_ocr = a.ocr == "on" or (a.ocr == "auto" and scanned >= max(1, round(n * 0.6))) if use_ocr and n > a.max_ocr_pages: fail("This looks like a scanned PDF with %d pages. Scans are read page by page, so the limit is %d pages — split it first." % (n, a.max_ocr_pages)) if use_ocr: try: ocr_to_docx(src, a.output, a.lang) except MemoryError: fail("This scan is too large to read in one go. Split it into smaller parts and try again.") except Exception as e: msg = str(e) if "tessdata" in msg.lower() or "language" in msg.lower(): fail("Text recognition is not available for that language.", 3, msg) fail("The scanned pages could not be read.", 3, repr(e)) src.close() say({"ok": True, "pages": n, "mode": "ocr", "scanned_pages": scanned, "lists_fixed": 0, "seconds": round(time.time() - t0, 2)}, 0) src.close() try: from pdf2docx import Converter cv = Converter(a.input) cv.convert(a.output, ignore_page_error=True, multi_processing=False) cv.close() except MemoryError: fail("This PDF is too complex to convert in one go. Split it into smaller parts and try again.") except Exception as e: fail("This PDF could not be converted.", 3, repr(e)) if not os.path.exists(a.output) or os.path.getsize(a.output) < 1000: fail("This PDF could not be converted.", 3, "empty output") try: lists_fixed = tidy_docx(a.output) except Exception: lists_fixed = 0 # the untouched pdf2docx output is still a valid file say({"ok": True, "pages": n, "mode": "layout", "scanned_pages": scanned, "lists_fixed": lists_fixed, "seconds": round(time.time() - t0, 2)}, 0) if __name__ == "__main__": main()