from __future__ import annotations
import ctypes
import os
from dataclasses import dataclass
from pathlib import Path
__all__ = [
"Canonical",
"FfiError",
"LibraryNotFoundError",
"find_cdylib",
"extract_canonical",
"parse_canonical",
"STATUS_OK",
"STATUS_INVALID",
"STATUS_REJECTED",
"STATUS_UNSUPPORTED",
]
STATUS_OK = 0
STATUS_INVALID = -1
STATUS_REJECTED = -2
STATUS_UNSUPPORTED = -3
CDYLIB_NAMES = ("pith_pdf.dll", "libpith_pdf.so", "libpith_pdf.dylib")
@dataclass(frozen=True)
class Canonical:
pages: int
rebuilt: bool
text: str
raw: bytes
@property
def text_bytes(self) -> bytes:
return self.raw[13:]
class LibraryNotFoundError(OSError):
class FfiError(Exception):
def __init__(self, op: str, status: int) -> None:
kind = {
STATUS_INVALID: "invalid argument",
STATUS_REJECTED: "input rejected",
STATUS_UNSUPPORTED: "unsupported feature",
}.get(status, "unknown failure")
super().__init__(f"{op} failed: {kind} (status {status})")
self.status = status
def find_cdylib() -> Path:
explicit = os.environ.get("PITH_CDYLIB")
if explicit:
p = Path(explicit)
if p.is_file():
return p
env_dir = os.environ.get("PITH_CDYLIB_DIR")
candidates: list[Path] = []
if env_dir:
env_dir_path = Path(env_dir)
candidates.append(env_dir_path)
if not env_dir_path.is_absolute():
candidates.append(Path.cwd() / env_dir_path)
candidates.append(Path(__file__).resolve().parents[3] / env_dir_path)
candidates.append(Path(__file__).resolve().parent) candidates.append(Path(__file__).resolve().parents[3] / "target" / "release")
for directory in candidates:
for name in CDYLIB_NAMES:
p = directory / name
if p.is_file():
return p
raise LibraryNotFoundError(
"no pith-pdf cdylib found (searched PITH_CDYLIB, PITH_CDYLIB_DIR, "
"the package directory and <repo>/target/release); "
"run `cargo build --release` first"
)
_lib: ctypes.CDLL | None = None
def _load() -> ctypes.CDLL:
global _lib
if _lib is None:
lib = ctypes.CDLL(str(find_cdylib()))
lib.pith_pdf_extract_canonical.argtypes = [
ctypes.c_void_p, ctypes.c_size_t, ctypes.POINTER(ctypes.c_void_p), ctypes.POINTER(ctypes.c_size_t), ]
lib.pith_pdf_extract_canonical.restype = ctypes.c_int32
lib.pith_pdf_free.argtypes = [ctypes.c_void_p, ctypes.c_size_t]
lib.pith_pdf_free.restype = None
_lib = lib
return _lib
def extract_canonical(data: bytes) -> bytes:
out = ctypes.c_void_p()
out_len = ctypes.c_size_t()
status = _load().pith_pdf_extract_canonical(data, len(data), ctypes.byref(out), ctypes.byref(out_len))
if status != STATUS_OK:
raise FfiError("pith_pdf_extract_canonical", status)
try:
return ctypes.string_at(out, out_len.value)
finally:
_load().pith_pdf_free(out, out_len.value)
def parse_canonical(raw: bytes) -> Canonical:
if len(raw) < 13:
raise ValueError("canonical stream is shorter than the 13-byte header")
text_len = int.from_bytes(raw[5:13], "big")
if len(raw) < 13 + text_len:
raise ValueError("canonical stream is shorter than its declared text length")
return Canonical(
pages=int.from_bytes(raw[0:4], "big"),
rebuilt=raw[4] == 1,
text=raw[13 : 13 + text_len].decode("utf-8"),
raw=raw,
)