SimplyParseDocs

Process a folder of documents

Submit a folder of PDFs and images with the async API, wait for every result, and save them as JSON Lines.

This script processes every supported file in a folder:

  • submits files with the async endpoint, a few at a time
  • records each document_id in a state file, so a re-run skips files already submitted instead of paying for them twice
  • polls until each document is completed or failed
  • writes one line per entry to results.jsonl

Run it

export SIMPLYPARSE_API_TOKEN="your-token"
export PARSER_SLUG="a1b2c3"

pip install requests
python batch.py ./invoices

Re-run the same command after an interruption; it carries on where it stopped.

The script

batch.py
"""Process every PDF/image in a folder with SimplyParse and save the results."""

import json
import os
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path

import requests

API = "https://api.simplyparse.com/dapi/v1/parser"
SLUG = os.environ["PARSER_SLUG"]
HEADERS = {"Authorization": f"Token {os.environ['SIMPLYPARSE_API_TOKEN']}"}
EXTENSIONS = {".pdf", ".png", ".jpg", ".jpeg", ".webp"}
CONCURRENCY = 4  # parallel uploads
STATE_FILE = Path("batch-state.json")  # file name -> document_id
RESULTS_FILE = Path("results.jsonl")


def load_state() -> dict:
    return json.loads(STATE_FILE.read_text()) if STATE_FILE.exists() else {}


def save_state(state: dict) -> None:
    STATE_FILE.write_text(json.dumps(state, indent=2))


def submit(path: Path) -> str:
    """Queue one file. Returns its document_id."""
    content = path.read_bytes()  # bytes, so a retry can resend them
    for attempt in range(1, 5):
        try:
            response = requests.post(
                f"{API}/{SLUG}/parse/async",
                headers=HEADERS,
                files={"file": (path.name, content)},
                data={"environment": "batch"},
                timeout=120,
            )
            if response.status_code < 500:
                body = response.json()
                if body.get("status") == "success":
                    return body["data"]["document_id"]
                if body.get("code") != "file_upload_error":
                    raise RuntimeError(f"{path.name}: {body.get('code')} {body.get('message') or body.get('detail')}")
        except (requests.ConnectionError, requests.Timeout):
            pass  # transient: retry below
        time.sleep(2**attempt)
    raise RuntimeError(f"{path.name}: gave up after retries")


def wait(document_id: str, max_wait: float = 900) -> dict:
    """Poll until the document is completed or failed. Returns `data`."""
    deadline = time.monotonic() + max_wait
    delay = 2.0
    while True:
        response = requests.get(f"{API}/{SLUG}/document/{document_id}", headers=HEADERS, timeout=30)
        body = response.json()
        if body.get("status") != "success":
            raise RuntimeError(f"{document_id}: {body.get('code')} {body.get('message') or body.get('detail')}")
        if body["data"]["status"] in ("completed", "failed"):
            return body["data"]
        if time.monotonic() > deadline:
            raise TimeoutError(f"{document_id} still {body['data']['status']}")
        time.sleep(delay)
        delay = min(delay * 1.5, 30)


def main(folder: str) -> None:
    files = sorted(p for p in Path(folder).iterdir() if p.suffix.lower() in EXTENSIONS)
    state = load_state()
    pending = [p for p in files if p.name not in state]
    print(f"{len(files)} files, {len(files) - len(pending)} already submitted")

    # 1. Submit new files, a few at a time. Save state after each one.
    with ThreadPoolExecutor(CONCURRENCY) as pool:
        for path, result in zip(pending, pool.map(lambda p: _try(submit, p), pending)):
            if isinstance(result, Exception):
                print(f"  ✗ {result}")
                continue
            state[path.name] = result
            save_state(state)
            print(f"  ↑ {path.name} → {result}")

    # 2. Wait for every submitted document and record its entries.
    done = {json.loads(line)["file"] for line in RESULTS_FILE.open()} if RESULTS_FILE.exists() else set()
    with RESULTS_FILE.open("a") as out:
        for name, document_id in state.items():
            if name in done:
                continue
            data = _try(wait, document_id)
            if isinstance(data, Exception):
                print(f"  ✗ {name}: {data}")
                continue
            for entry in data["entries"] or [{}]:
                out.write(json.dumps({
                    "file": name,
                    "document_id": document_id,
                    "document_status": data["status"],
                    "entry_id": entry.get("id"),
                    "is_valid": entry.get("is_valid"),
                    "parsed_data": entry.get("parsed_data"),
                    "validation_errors": entry.get("validation_errors"),
                }) + "\n")
            print(f"  ✓ {name}: {data['status']}, {len(data['entries'])} entr{'y' if len(data['entries']) == 1 else 'ies'}")


def _try(fn, arg):
    try:
        return fn(arg)
    except Exception as exc:  # report and carry on with the rest
        return exc


if __name__ == "__main__":
    main(sys.argv[1] if len(sys.argv) > 1 else ".")

Adapt it

  • Send to a database instead of a file. Replace the out.write(...) block with an upsert keyed on entry_id.
  • Skip polling. Add a webhook for the batch environment and drop step 2. See Receive results with a webhook.
  • Watch your balance. Once credits run out, further submissions fail with insufficient_balance. Top up and re-run; files already submitted are skipped.
  • Tune CONCURRENCY. Four parallel uploads is a sensible start. Raise it gradually for large batches.

On this page