#!/usr/bin/env python3 import json import mimetypes import os import sys import time import urllib.error import urllib.request def main(): base = os.environ["PDFSCRIBE_API_BASE"].rstrip("/") key, filename = os.environ["PDFSCRIBE_API_KEY"], os.environ["PDFSCRIBE_FILE"] timeout = int(os.environ.get("PDFSCRIBE_POLL_TIMEOUT_SECONDS", "300")) if not 1 <= timeout <= 3600: raise SystemExit("PDFSCRIBE_POLL_TIMEOUT_SECONDS must be 1 to 3600") def request(path, method="GET", body=None, headers=None, limit=30): data = json.dumps(body).encode() if body is not None else None fields = {"Authorization": "Bearer " + key, **(headers or {})} if body is not None: fields["Content-Type"] = "application/json" with urllib.request.urlopen(urllib.request.Request(base + path, data=data, headers=fields, method=method), timeout=limit) as response: return json.loads(response.read() or b"{}") content_type = os.environ.get("PDFSCRIBE_CONTENT_TYPE") or mimetypes.guess_type(filename)[0] or "application/pdf" upload_key = "quickstart-upload-" + str(time.time_ns()) upload = request("/uploads", "POST", {"content_type": content_type, "size_bytes": os.path.getsize(filename)}, {"Idempotency-Key": upload_key}) with open(filename, "rb") as source: signed = urllib.request.Request(upload["upload"]["url"], data=source.read(), headers={k: ",".join(v) for k, v in upload["upload"]["headers"].items()}, method=upload["upload"]["method"]) with urllib.request.urlopen(signed, timeout=30) as response: response.read() request("/uploads/" + upload["id"] + "/complete", "POST") job = request("/extractions", "POST", {"upload_id": upload["id"], "mode": "ocr", "pages": "all", "output": {"layout": "ordered", "include": ["text", "tables"], "table_format": "markdown", "confidence": "block"}}, {"Idempotency-Key": "quickstart-" + str(time.time_ns())}) print("Accepted extraction ID: " + job["id"], file=sys.stderr) deadline = time.monotonic() + timeout while job["status"] not in ("succeeded", "partially_succeeded", "failed", "cancelled", "expired"): remaining = deadline - time.monotonic() if remaining <= 0: raise SystemExit("Polling timed out; resume checking the existing extraction ID: " + job["id"]) try: job = request("/extractions/" + job["id"], limit=min(30, remaining)) except (TimeoutError, urllib.error.URLError) as error: timed_out = isinstance(error, TimeoutError) or isinstance(getattr(error, "reason", None), TimeoutError) if remaining <= 30 and timed_out and time.monotonic() >= deadline: raise SystemExit("Polling timed out; resume checking the existing extraction ID: " + job["id"]) raise if job["status"] not in ("succeeded", "partially_succeeded", "failed", "cancelled", "expired"): time.sleep(min(1, max(0, deadline - time.monotonic()))) if job["status"] not in ("succeeded", "partially_succeeded"): raise SystemExit("Extraction ended with status: " + job["status"]) for path in ("/extractions/" + job["id"] + "/result?view=ordered&include=text,tables&table_format=markdown&confidence=block", "/credits", "/usage?page=1"): print(json.dumps(request(path), indent=2)) if __name__ == "__main__": try: main() except Exception: # Transport errors can contain signed URLs. Never print exception details. raise SystemExit("Quickstart request failed; check connectivity and the existing extraction before retrying.")