Make any PDF readable for AI
Drop a PDF and get clean, structured text back — even from scanned pages. Then split it into bite-sized pieces, ready for a knowledge base.
- 1 Upload
- 2 Read
- 3 Split
Your document, as text
How it works
Read
AIVAX pulls the text out of your PDF and keeps its structure — headings, lists, tables. Pages that are just images get read with OCR. Fetch & OCR
Split
A model finds where one idea ends and the next begins, so each chunk can answer a question on its own. Your text is never rewritten. Text segmentation
Search
Load the chunks into a collection and your assistant can find and quote the right passage of your document. Collections
For developers API calls from this session, and the code to build it yourself
Calls made by this page
- API calls
- 0
- Processing units
- 0
- Left this minute
- –
Estimates use public list prices before daily allowances. Limits on this toy: 5 documents per minute, 20 per hour. A Turnstile check runs before every call.
Build it yourself
Two API calls: POST /api/v1/web/fetch and POST /api/v1/generations/segment.
See the API reference or the
Worker behind this page.
# 1. PDF → Markdown. Inline the file as a base64 data URI (10 MB max per item). curl https://inference.aivax.net/api/v1/web/fetch \ -H "Authorization: Bearer $AIVAX_API_KEY" \ -H "Content-Type: application/json" \ -d "{ \"contents\": [\"data:application/pdf;base64,$(base64 -w0 manual.pdf)\"], \"returnErrors\": true }" # → { "data": { "results": [ { # "index": 0, # "extractedText": "# Title\n\nParagraph...", ← Markdown # "processingUnits": 7, ← what you pay for # "jsonProcessingUnits": 0, # "error": null } ] } } # 2. Markdown → segments. `sanitize` skips fragments that are useless for retrieval. curl https://inference.aivax.net/api/v1/generations/segment \ -H "Authorization: Bearer $AIVAX_API_KEY" \ -H "Content-Type: application/json" \ -d '{ "documents": ["# Title\n\nParagraph..."], "sanitize": true }' # → { "data": { # "result": [ { "index": 0, "count": 8, "segments": ["...", "..."] } ], # "usage": { "processing_units": 2516, "cost": 0 } } }
// Works in Node 18+, Bun, Deno and browsers. Keep the API key server-side. const AIVAX = "https://inference.aivax.net/api/v1"; const headers = { authorization: `Bearer ${process.env.AIVAX_API_KEY}`, "content-type": "application/json", }; // 1. PDF → Markdown async function pdfToMarkdown(bytes) { const base64 = Buffer.from(bytes).toString("base64"); const response = await fetch(`${AIVAX}/web/fetch`, { method: "POST", headers, body: JSON.stringify({ contents: [`data:application/pdf;base64,${base64}`], returnErrors: true, }), }); const { data } = await response.json(); const [item] = data.results; // one result per `contents` entry, matched by index if (item.error) throw new Error(item.error); return { markdown: item.extractedText, processingUnits: item.processingUnits }; } // 2. Markdown → RAG segments async function segmentForRag(markdown, { sanitize = true } = {}) { const response = await fetch(`${AIVAX}/generations/segment`, { method: "POST", headers, body: JSON.stringify({ documents: [markdown], sanitize }), }); const { data } = await response.json(); return { segments: data.result[0].segments, processingUnits: data.usage.processing_units }; } // 3. Shape the segments as JSONL for POST /api/v1/collections/{id}/documents function toJsonl(segments, source) { return segments .map((text, index) => JSON.stringify({ docid: `${source}#${index + 1}`, // stable name: re-importing updates instead of duplicating text, __ref: source.slice(0, 64), // groups the chunks of one file __tags: ["pdf"], __meta: { source, index }, })) .join("\n"); } const bytes = await Bun.file("manual.pdf").arrayBuffer(); const { markdown } = await pdfToMarkdown(bytes); const { segments } = await segmentForRag(markdown); await Bun.write("manual.jsonl", toJsonl(segments, "manual.pdf"));
// Trimmed from src/worker.js in the repository. The real file adds Turnstile // verification and a Durable Object sliding-window rate limiter. const AIVAX_API = "https://inference.aivax.net/api/v1"; export default { async fetch(request, env) { const url = new URL(request.url); if (url.pathname === "/api/convert" && request.method === "POST") { const form = await request.formData(); const file = form.get("file"); const gate = await verifyTurnstileAndQuota(request, env, form.get("turnstile")); if (gate instanceof Response) return gate; // 403 or 429 const base64 = bytesToBase64(new Uint8Array(await file.arrayBuffer())); const call = await callAivax(env, "/web/fetch", { contents: [`data:application/pdf;base64,${base64}`], returnErrors: true, }); const item = call.data.results[0]; return Response.json({ markdown: item.extractedText, processingUnits: item.processingUnits, elapsedMs: call.elapsedMs, requestId: call.requestId, }); } return env.ASSETS.fetch(request); // static site from ./public }, }; async function callAivax(env, path, body) { const startedAt = Date.now(); const response = await fetch(AIVAX_API + path, { method: "POST", headers: { authorization: `Bearer ${env.AIVAX_API_KEY}`, // `wrangler secret put AIVAX_API_KEY` "content-type": "application/json", }, body: JSON.stringify(body), }); const payload = await response.json(); return { data: payload.data, elapsedMs: Date.now() - startedAt, requestId: response.headers.get("x-request-id"), }; }