Python SDK for the ProofPudding API - Document processing and Q&A
Project description
Proofpudding SDK
A Python SDK for the Proofpudding API - Document processing and question answering.
Installation
pip install proofpudding
Requirements
- Python 3.10+
- httpx >= 0.25.0
- pydantic >= 2.0
Quick Start
from pudding import PuddingClient
from pudding.exceptions import NotFoundError, ValidationError
# Initialize client
client = PuddingClient(access_token="pk_your_api_key")
# Check API health
health = client.health.check()
print(f"API Status: {health.status}")
# Upload a document (.pdf or .docx, max 100MB)
doc = client.documents.upload(file_path="./contract.pdf")
print(f"Uploaded: {doc.id}")
# Ask a question about the document
job = client.jobs.create(
document_id=doc.id,
question="What is the effective date of this contract?"
)
if job.success:
print(f"Answer: {job.result.answer}")
print(f"Confidence: {job.result.confidence}")
for citation in job.result.citations:
print(f" - Page {citation.page}: {citation.quote}")
else:
print(f"Failed: {job.error}")
# List all documents with pagination
docs = client.documents.list(skip=0, limit=10)
print(f"Total documents: {docs.total}")
for doc in docs.items:
print(f" - {doc.filename} ({doc.size_bytes} bytes)")
# List jobs for a specific document
jobs = client.jobs.list(document_id=doc.id)
# Delete a document (also deletes associated jobs)
try:
deleted = client.documents.delete(document_id=doc.id)
print(f"Deleted: {deleted.filename}")
except NotFoundError:
print("Document not found")
Async Support
The SDK provides both synchronous and asynchronous clients:
from pudding import PuddingClient, AsyncPuddingClient
# Synchronous client
client = PuddingClient(access_token="pk_your_api_key")
docs = client.documents.list()
# Asynchronous client
async_client = AsyncPuddingClient(access_token="pk_your_api_key")
docs = await async_client.documents.list()
Context Manager Support
Both clients support context managers for proper resource cleanup:
# Synchronous
with PuddingClient(access_token="pk_your_api_key") as client:
docs = client.documents.list()
# Asynchronous
async with AsyncPuddingClient(access_token="pk_your_api_key") as client:
docs = await client.documents.list()
Configuration
Initialization Options
client = PuddingClient(
access_token="pk_your_api_key", # Required: Your API key
timeout=1800.0, # Optional: Request timeout in seconds (default: 1800)
max_retries=3, # Optional: Max retries for transient errors (default: 3)
)
Custom Timeout for Jobs
Job creation is a blocking operation that waits for document processing. The default timeout is 1800 seconds (30 minutes):
# Use a custom timeout for the entire client
client = PuddingClient(access_token="...", timeout=3600.0) # 1 hour
# Or override per-call
job = client.jobs.create(
document_id="uuid-string",
question="...",
timeout=900.0, # 15 minutes for this call only
)
API Reference
Health Checks
# Check API health
health = client.health.check()
print(health.status) # "healthy"
print(health.version) # API version
print(health.environment) # Environment name
# Check readiness (includes database status)
ready = client.health.ready()
print(ready.status) # "ready" or "not_ready"
print(ready.database) # "connected" or error message
Documents
Supported file types: .pdf, .docx (max 100MB).
# Upload a document from file path
doc = client.documents.upload(file_path="/path/to/document.pdf")
# Upload a .docx file
doc = client.documents.upload(file_path="/path/to/report.docx")
# Upload a document from bytes
with open("document.pdf", "rb") as f:
doc = client.documents.upload(file=f.read(), filename="document.pdf")
# List documents with pagination
doc_list = client.documents.list(skip=0, limit=20)
print(f"Total: {doc_list.total}")
for doc in doc_list.items:
print(f"{doc.id}: {doc.filename}")
# Delete a document
deleted_doc = client.documents.delete(document_id="uuid-string")
Jobs
# Create a job (process document with a question)
job = client.jobs.create(
document_id="uuid-string",
question="What is the total revenue mentioned in this document?"
)
# Access job results
if job.success:
print(job.result.answer)
print(job.result.confidence) # "high", "medium", "low", or "not_found"
print(job.processing_time_ms) # Processing duration in ms
for citation in job.result.citations:
print(f"Page {citation.page}: {citation.quote}")
# Cost information (if available)
if job.usage:
print(f"Cost: {job.usage.total_cost_cents} cents")
# List all jobs
job_list = client.jobs.list(skip=0, limit=20)
# List jobs for a specific document
job_list = client.jobs.list(document_id="uuid-string")
Structured Output
Use JobConfig with OutputSchemaConfig to get structured JSON output:
from pudding.models import JobConfig, OutputSchemaConfig
config = JobConfig(
output_schema=OutputSchemaConfig(
schema={
"type": "object",
"properties": {
"effective_date": {"type": "string"},
"parties": {"type": "array", "items": {"type": "string"}},
},
},
strict=True, # Fail if schema can't be satisfied (default: True)
include_citations=True, # Include citations in response (default: True)
include_raw_answer=False, # Include raw text answer (default: False)
),
verify_citations=True, # Verify citations against source (default: True)
reasoning_effort="auto", # "auto", "low", or "high" (default: "auto")
)
job = client.jobs.create(
document_id="uuid-string",
question="Extract the effective date and parties",
config=config,
)
if job.success and job.result.structured_output:
print(job.result.structured_output)
# {"effective_date": "2025-01-01", "parties": ["Acme Corp", "Widget Inc"]}
Streaming
Use create_stream for real-time progress updates via Server-Sent Events:
from pudding.models import (
DownloadingEvent,
ProcessingEvent,
ThinkingEvent,
StepEvent,
VerifyingEvent,
CompleteEvent,
ErrorEvent,
)
# Async streaming
async for event in client.jobs.create_stream(
document_id="uuid-string",
question="What is the effective date?"
):
if isinstance(event, DownloadingEvent):
print(f"Downloading: {event.progress}%")
elif isinstance(event, ProcessingEvent):
print(f"Processing: {event.pages_done}/{event.total_pages}")
elif isinstance(event, ThinkingEvent):
print(f"Thinking (iteration {event.iteration})...")
elif isinstance(event, StepEvent):
print(f"Step: {event.message}")
elif isinstance(event, VerifyingEvent):
print(f"Verifying citations...")
elif isinstance(event, CompleteEvent):
print(f"Done: {event.result}")
elif isinstance(event, ErrorEvent):
print(f"Error [{event.error_code}]: {event.message}")
Exception Handling
The SDK provides a comprehensive exception hierarchy:
from pudding.exceptions import (
PuddingError, # Base exception for all SDK errors
AuthenticationError, # 401: Invalid or missing API key
NotFoundError, # 404: Resource not found
ValidationError, # 400: Invalid request
RateLimitError, # 429: Too many requests
ServerError, # 500: Internal server error
GatewayError, # 502: Upstream service error
TimeoutError, # 504: Request timed out
)
try:
doc = client.documents.upload(file_path="./document.txt")
except ValidationError as e:
print(f"Validation failed: {e.message}") # "Only .pdf, .docx files are allowed"
except AuthenticationError as e:
print(f"Auth failed: {e.message}")
except PuddingError as e:
print(f"API error ({e.status_code}): {e.message}")
Logging
The SDK uses Python's standard logging module. Configure logging level as needed:
import logging
# Enable debug logging for the SDK
logging.getLogger("pudding").setLevel(logging.DEBUG)
Project details
Download files
Download the file for your platform. If you're not sure which to choose, learn more about installing packages.
Source Distribution
Built Distribution
Filter files by name, interpreter, ABI, and platform.
If you're not sure about the file name format, learn more about wheel file names.
Copy a direct link to the current filters
File details
Details for the file proofpudding-0.1.6.tar.gz.
File metadata
- Download URL: proofpudding-0.1.6.tar.gz
- Upload date:
- Size: 15.6 kB
- Tags: Source
- Uploaded using Trusted Publishing? No
- Uploaded via: twine/6.2.0 CPython/3.12.3
File hashes
| Algorithm | Hash digest | |
|---|---|---|
| SHA256 |
9d3687a2f045b8a327eff55ac8c279a00575c561190ec950aeefa4f508f88a8c
|
|
| MD5 |
842c291e120fe0ccf4c257cb9748a91f
|
|
| BLAKE2b-256 |
d32a8c125f58f85720bbcb67080146032d657327e2efbd6746e47665bd51edbd
|
File details
Details for the file proofpudding-0.1.6-py3-none-any.whl.
File metadata
- Download URL: proofpudding-0.1.6-py3-none-any.whl
- Upload date:
- Size: 29.1 kB
- Tags: Python 3
- Uploaded using Trusted Publishing? No
- Uploaded via: twine/6.2.0 CPython/3.12.3
File hashes
| Algorithm | Hash digest | |
|---|---|---|
| SHA256 |
4735d3a16f472fb4166c41842a9bab73944de2a7f6543817c2dc58efbab27e13
|
|
| MD5 |
186f03078e88e19c00c2e8aa321f3d6d
|
|
| BLAKE2b-256 |
f014a83b7050496f6dfcc3cf2c286057a7f345b11fd6bd7dff33c6f11f4931c7
|