Python PDF to Markdown
Convert a selected folder of local PDFs sequentially, save Markdown safely and recover with the same inputs and identities.
Use Python 3 with requests for local-file integration. Start with 1–5 finished PDFs in a directory you explicitly select. The recipe converts them sequentially, with a three-page cap per PDF, and saves one Markdown file per input.
Convert a small folder of PDFs
In a virtual environment, run pip install requests. Set PDFTOMARKDOWN_API_KEY to your account key. Save the code as batch.py, then run python batch.py selected-pdfs batch-state. Choose an empty state directory for this new batch; run one process at a time. Requires Linux or macOS.
The default selects at most three pages from each PDF. Set OPTIONS before creating a new batch if you need another cap. New accounts receive 20 trial pages once; further conversions use paid credits. Each delivered page uses one credit. Files go to an external processor; review processing and retention first.
Show batch.py and copy the complete recipe
import base64
import hashlib
import json
import os
import sys
import tempfile
import time
import uuid
from pathlib import Path
import requests
# Usage: python batch.py selected-pdfs batch-state [--resume]
# Keep batch-state: it contains private PDF snapshots and saved identities.
ENDPOINT = "https://pdftomarkdown.dev/v1/convert"
MAX_FILES = 5
OPTIONS = {"max_pages": 3, "include_raw": False}
RECOVERY = ("Stopped. Keep batch-state and its immutable inputs/options/identities. "
"Read /docs/#errors before explicitly rerunning with --resume. "
"Never delete state to retry; after replay expiry, reconcile manually.")
def durable_write(path, data, replace=False):
# A complete file becomes visible atomically; new outputs never overwrite.
fd, temporary = tempfile.mkstemp(dir=path.parent)
try:
with os.fdopen(fd, "wb") as output:
output.write(data)
output.flush()
os.fsync(output.fileno())
if replace:
os.replace(temporary, path)
else:
os.link(temporary, path)
directory = os.open(path.parent, os.O_RDONLY)
try:
os.fsync(directory)
finally:
os.close(directory)
finally:
if os.path.exists(temporary):
os.unlink(temporary)
def main():
if len(sys.argv) not in (3, 4) or (len(sys.argv) == 4 and sys.argv[3] != "--resume"):
sys.exit("Usage: python batch.py selected-pdfs batch-state [--resume]")
key = os.environ.get("PDFTOMARKDOWN_API_KEY")
if not key or key == "demo_public_key":
sys.exit("Set PDFTOMARKDOWN_API_KEY to your account key")
selected, state = (Path(value).resolve() for value in sys.argv[1:3])
resume = len(sys.argv) == 4
state.mkdir(mode=0o700, parents=True, exist_ok=True)
lock = state / "running.lock"
try:
lock.mkdir()
except FileExistsError:
sys.exit("Batch locked. Confirm no process is running before removing running.lock. " + RECOVERY)
try:
manifest = state / "manifest.json"
if manifest.exists():
batch = json.loads(manifest.read_text(encoding="utf-8"))
if batch["selected"] != str(selected):
sys.exit("This state belongs to another selected directory. " + RECOVERY)
else:
if resume:
sys.exit("Recovery state is missing. " + RECOVERY)
files = sorted(selected.glob("*.pdf"))
if not selected.is_dir() or not 1 <= len(files) <= MAX_FILES:
sys.exit("Select a directory containing 1 to 5 finished .pdf files")
jobs = []
for pdf in files:
if not pdf.is_file() or not 0 < pdf.stat().st_size <= 50 * 1024 * 1024:
sys.exit("Each selected PDF must be between 1 byte and 50 MiB")
data = pdf.read_bytes()
jobs.append({"output": pdf.name + ".md", "identity": str(uuid.uuid4()),
"sha256": hashlib.sha256(data).hexdigest(),
"input": {"pdf_base64": base64.b64encode(data).decode("ascii"), **OPTIONS},
"attempted_at": None, "result": None})
batch = {"selected": str(selected), "jobs": jobs}
durable_write(manifest, json.dumps(batch).encode("utf-8"))
def save():
durable_write(manifest, json.dumps(batch).encode("utf-8"), replace=True)
# Check every destination and snapshot before submitting any work.
for job in batch["jobs"]:
data = base64.b64decode(job["input"]["pdf_base64"], validate=True)
if hashlib.sha256(data).hexdigest() != job["sha256"]:
sys.exit("Saved input is damaged. " + RECOVERY)
output = state / job["output"]
if output.exists() and (job["result"] is None or
output.read_bytes() != job["result"]["markdown"].encode("utf-8")):
sys.exit("Output already exists; no file was overwritten. " + RECOVERY)
with requests.Session() as session:
for job in batch["jobs"]:
output = state / job["output"]
if job["result"] is None:
if job["attempted_at"] is not None:
age = time.time() - job["attempted_at"]
if not resume or age < 0 or age >= 24 * 60 * 60:
sys.exit("Explicit recovery within the 24-hour window is required. " + RECOVERY)
else:
job["attempted_at"] = time.time()
save() # Durable identity, exact input/options and attempt time before POST.
response = session.post(
ENDPOINT,
headers={"Authorization": f"Bearer {key}",
"Idempotency-Key": job["identity"]},
json={"input": job["input"]}, timeout=(10, 660),
allow_redirects=False,
)
if response.status_code != 200:
sys.exit(f"HTTP {response.status_code}. " + RECOVERY)
result = response.json()
if not (isinstance(result, dict) and result.get("complete") is True
and isinstance(result.get("markdown"), str)
and type(result.get("pages")) is int and result["pages"] >= 0
and isinstance(result.get("request_id"), str) and result["request_id"].strip()):
sys.exit("Incomplete or invalid conversion response. " + RECOVERY)
job["result"] = result
save() # Recovery can finish writing without another conversion.
if not output.exists():
durable_write(output, job["result"]["markdown"].encode("utf-8"))
print("Batch complete. Markdown saved in batch-state; retain its manifest.")
finally:
lock.rmdir()
try:
main()
except (requests.RequestException, ValueError, OSError, KeyError, TypeError):
sys.exit(RECOVERY)Recover a stopped batch
Keep batch-state private and intact. Its manifest stores PDF snapshots, options, identities and completed responses before writing filename.pdf.md. Subsequent runs use those saved snapshots, even if you edit the original directory or the script’s options. Do not edit the manifest or reuse this state directory for a different batch.
HTTP 401, 429, 409, malformed output and connection failures stop the batch. There is no automatic retry or purchase. For a rate limit or pending request, follow error and retry guidance before running python batch.py selected-pdfs batch-state --resume. A timeout stops waiting; processing and charging may continue.
Completed account results replay for 24 hours. This recipe conservatively blocks resubmission 24 hours after its first attempt, including when the outcome is unknown. After expiry, reconcile the earlier request before deliberately starting any new conversion; never delete the manifest to make a retry work. If a crash leaves running.lock, confirm the old process has stopped before removing only that lock.
Use the saved Markdown
Simple tables are escaped GFM; complex tables may contain sanitized HTML. Use the table extraction workflow to parse a saved file and handle the HTML branch explicitly. OCR does not produce validated invoice fields. For tool selection, see Python PDF parsing.
Single-request transports
These smaller examples use Python 3 and requests. Set PDFTOMARKDOWN_API_KEY and save a unique PDFTOMARKDOWN_IDEMPOTENCY_KEY before submission. Reuse that identity only with identical document bytes and options.
Allow up to 11 minutes for a synchronous request. Stopping the client or reaching its timeout stops waiting; processing and charging may continue. Measure elapsed time locally; optional provider timing headers do not measure total request time.
Python URL
import base64
import os
import sys
from pathlib import Path
import requests
key = os.environ.get("PDFTOMARKDOWN_API_KEY")
identity = os.environ.get("PDFTOMARKDOWN_IDEMPOTENCY_KEY")
if not key or not identity:
sys.exit("Set PDFTOMARKDOWN_API_KEY and a saved PDFTOMARKDOWN_IDEMPOTENCY_KEY")
try:
response = requests.post(
"https://pdftomarkdown.dev/v1/convert",
headers={"Authorization": f"Bearer {key}", "Idempotency-Key": identity,
"Content-Type": "application/json"},
json={"input": {"pdf_url": "https://pdftomarkdown.dev/samples/invoice.pdf"}},
timeout=(10, 660),
)
if response.status_code != 200:
sys.exit(f"HTTP {response.status_code}; keep the identity for recovery")
result = response.json()
if not (isinstance(result, dict) and result.get("complete") is True
and isinstance(result.get("markdown"), str)
and type(result.get("pages")) is int and result["pages"] >= 0
and isinstance(result.get("request_id"), str) and result["request_id"].strip()):
sys.exit("Incomplete or invalid conversion response")
sys.stdout.write(result["markdown"])
except (requests.RequestException, ValueError, OSError):
sys.exit("Conversion failed; keep the identity and original input for recovery")Local PDF as Base64
import base64
import os
import sys
from pathlib import Path
import requests
key = os.environ.get("PDFTOMARKDOWN_API_KEY")
identity = os.environ.get("PDFTOMARKDOWN_IDEMPOTENCY_KEY")
if not key or not identity:
sys.exit("Set PDFTOMARKDOWN_API_KEY and a saved PDFTOMARKDOWN_IDEMPOTENCY_KEY")
try:
response = requests.post(
"https://pdftomarkdown.dev/v1/convert",
headers={"Authorization": f"Bearer {key}", "Idempotency-Key": identity,
"Content-Type": "application/json"},
json={"input": {"pdf_base64": base64.b64encode(Path("document.pdf").read_bytes()).decode("ascii")}},
timeout=(10, 660),
)
if response.status_code != 200:
sys.exit(f"HTTP {response.status_code}; keep the identity for recovery")
result = response.json()
if not (isinstance(result, dict) and result.get("complete") is True
and isinstance(result.get("markdown"), str)
and type(result.get("pages")) is int and result["pages"] >= 0
and isinstance(result.get("request_id"), str) and result["request_id"].strip()):
sys.exit("Incomplete or invalid conversion response")
sys.stdout.write(result["markdown"])
except (requests.RequestException, ValueError, OSError):
sys.exit("Conversion failed; keep the identity and original input for recovery")Raw PDF upload
import base64
import os
import sys
from pathlib import Path
import requests
key = os.environ.get("PDFTOMARKDOWN_API_KEY")
identity = os.environ.get("PDFTOMARKDOWN_IDEMPOTENCY_KEY")
if not key or not identity:
sys.exit("Set PDFTOMARKDOWN_API_KEY and a saved PDFTOMARKDOWN_IDEMPOTENCY_KEY")
try:
response = requests.post(
"https://pdftomarkdown.dev/v1/convert",
headers={"Authorization": f"Bearer {key}", "Idempotency-Key": identity,
"Content-Type": "application/pdf"},
data=Path("document.pdf").read_bytes(),
timeout=(10, 660),
)
if response.status_code != 200:
sys.exit(f"HTTP {response.status_code}; keep the identity for recovery")
result = response.json()
if not (isinstance(result, dict) and result.get("complete") is True
and isinstance(result.get("markdown"), str)
and type(result.get("pages")) is int and result["pages"] >= 0
and isinstance(result.get("request_id"), str) and result["request_id"].strip()):
sys.exit("Incomplete or invalid conversion response")
sys.stdout.write(result["markdown"])
except (requests.RequestException, ValueError, OSError):
sys.exit("Conversion failed; keep the identity and original input for recovery")Only a complete, typed success writes Markdown. For pending work, rate limits or an uncertain connection, follow the API’s retry guidance using the same saved identity and input. Never replace the identity automatically. Completed account results replay for 24 hours; demo requests have no paid replay guarantee.
API reference: limits, authentication and errors · Document processing and retention