{"version":1,"pages":[{"id":"tQ8rQLPlTOMzF7HMgEIw","title":"Welcome","pathname":"/parasail-docs","siteSpaceId":"sitesp_Wgad2","description":"Parasail provides affordable, high-performance cloud GPUs for running demanding AI workloads—serverless inference, dedicated instances, and batch processing."},{"id":"cqVFBa1P7r5p7nfFbCnL","title":"Serverless","pathname":"/parasail-docs/quickstart/serverless","siteSpaceId":"sitesp_Wgad2","description":"Make your first serverless inference call in under two minutes using the OpenAI SDK.","breadcrumbs":[{"label":"Quickstart"}]},{"id":"UcIhEOaxPYRZMC3qQ7xw","title":"Dedicated Instances","pathname":"/parasail-docs/quickstart/dedicated","siteSpaceId":"sitesp_Wgad2","description":"Deploy your own model on a dedicated GPU instance in a few minutes.","breadcrumbs":[{"label":"Quickstart"}]},{"id":"OrPZ7ZRehYXxzse9LB3u","title":"Batch Processing","pathname":"/parasail-docs/quickstart/batch","siteSpaceId":"sitesp_Wgad2","description":"Submit your first batch job in five lines of Python. Batch processing is 50% off serverless pricing.","breadcrumbs":[{"label":"Quickstart"}]},{"id":"AbuAPWxU9zDfEIgukNsK","title":"Serverless","pathname":"/parasail-docs/products/overview","siteSpaceId":"sitesp_Wgad2","description":"Our Serverless Service offers API based on Token Usage with Popular Models.","breadcrumbs":[{"label":"Products"}]},{"id":"WqVfOfAeufKv6KvjqS2V","title":"Model-specific Notes","pathname":"/parasail-docs/products/overview/model-specific-notes","siteSpaceId":"sitesp_Wgad2","description":"Model-specific Serverless parameters for DeepSeek V3.1, Qwen3.5, and GPT-OSS models on Parasail.","breadcrumbs":[{"label":"Products"},{"label":"Serverless"}]},{"id":"2lUFwT4FsBP3efkJVjM2","title":"Responses API","pathname":"/parasail-docs/products/overview/responses-api","siteSpaceId":"sitesp_Wgad2","description":"Use the OpenAI Responses API for multi-turn, agentic workflows with built-in tool calling.","breadcrumbs":[{"label":"Products"},{"label":"Serverless"}]},{"id":"dvVsRgLfPYaM2uxDi8Gw","title":"Dedicated Instances","pathname":"/parasail-docs/products/overview-1","siteSpaceId":"sitesp_Wgad2","description":"Deploy any Hugging Face model on a private GPU endpoint with full control over hardware, scaling, and latency.","breadcrumbs":[{"label":"Products"}]},{"id":"RxY1ZB4dWvr4pDo23j6s","title":"Speculative Decoding","pathname":"/parasail-docs/products/overview-1/speculative-decoding","siteSpaceId":"sitesp_Wgad2","description":"Speed up dedicated models 1.5-2x by adding a draft model for speculative decoding on the same GPU.","breadcrumbs":[{"label":"Products"},{"label":"Dedicated Instances"}]},{"id":"0tnt4xWOMBgre4UPkN0j","title":"Private HuggingFace Models","pathname":"/parasail-docs/products/overview-1/private-hf-models","siteSpaceId":"sitesp_Wgad2","description":"How to create a dedicated deployment for a private HuggingFace model","breadcrumbs":[{"label":"Products"},{"label":"Dedicated Instances"}]},{"id":"iOuln9kZPt4Hu8ovY92W","title":"Management API","pathname":"/parasail-docs/products/overview-1/management-api","siteSpaceId":"sitesp_Wgad2","description":"Deploy, pause, resume, scale, and monitor dedicated endpoints programmatically with the Parasail REST API.","breadcrumbs":[{"label":"Products"},{"label":"Dedicated Instances"}]},{"id":"SaBR907K888ZJ6TL8eWh","title":"FP8 Quantization","pathname":"/parasail-docs/products/overview-1/fp8-quantization","siteSpaceId":"sitesp_Wgad2","description":"Quantize dedicated models to FP8 with llm-compressor to halve memory use and speed up inference with minimal accuracy loss.","breadcrumbs":[{"label":"Products"},{"label":"Dedicated Instances"}]},{"id":"NxkdYvHTxHMuQxFvDwid","title":"Auto-Scaling","pathname":"/parasail-docs/products/overview-1/auto-scaling","siteSpaceId":"sitesp_Wgad2","description":"Configure auto-scaling for dedicated endpoints using max concurrent requests, target concurrency, and smoothing factor.","breadcrumbs":[{"label":"Products"},{"label":"Dedicated Instances"}]},{"id":"7Z2IH4HIhx1WC35hiZ7X","title":"Dedicated Serverless","pathname":"/parasail-docs/products/capacity","siteSpaceId":"sitesp_Wgad2","description":"Understand the concurrency limits and max requests per minute that govern autoscaling capacity on Dedicated Serverless endpoints.","breadcrumbs":[{"label":"Products"}]},{"id":"51IF2LvdW8BHeIjbfIMj","title":"Batch","pathname":"/parasail-docs/products/quickstart","siteSpaceId":"sitesp_Wgad2","description":"Get started with Parasail's Batch Processing through the UI or the OpenAI-compatible Python batch helper library.","breadcrumbs":[{"label":"Products"}]},{"id":"WOGJGPpYEdkQ6pPXPjpJ","title":"Batch File Format","pathname":"/parasail-docs/products/quickstart/file-format","siteSpaceId":"sitesp_Wgad2","description":"Format batch input and output .jsonl files for Parasail's OpenAI-compatible Batch API.","breadcrumbs":[{"label":"Products"},{"label":"Batch"}]},{"id":"8VutuisCPLG1lVxlV0gs","title":"Troubleshooting","pathname":"/parasail-docs/products/quickstart/troubleshooting","siteSpaceId":"sitesp_Wgad2","description":"Diagnose and fix common batch job failures, from JSONL validation errors to quota and authentication issues.","breadcrumbs":[{"label":"Products"},{"label":"Batch"}]},{"id":"26CxcjJsL6rMJc99kXYY","title":"Image Generation","pathname":"/parasail-docs/products/overview-2","siteSpaceId":"sitesp_Wgad2","description":"Run diffusion models for batch image generation and editing with Parasail.","breadcrumbs":[{"label":"Products"}]},{"id":"4JPH773CpfUUTaMELFVR","title":"Authentication","pathname":"/parasail-docs/api-reference/authentication","siteSpaceId":"sitesp_Wgad2","description":"API key creation, base URLs, and authentication for the Parasail API.","breadcrumbs":[{"label":"API Reference"}]},{"id":"1bd04EH01Y93WGVrjw8x","title":"Chat Completions","pathname":"/parasail-docs/api-reference/chat-completions","siteSpaceId":"sitesp_Wgad2","description":"OpenAI-compatible chat completions API for serverless and dedicated model inference.","breadcrumbs":[{"label":"API Reference"}]},{"id":"1c9IBzqcRGEfyAUgb6Uy","title":"Responses API","pathname":"/parasail-docs/api-reference/responses-api","siteSpaceId":"sitesp_Wgad2","description":"Responses API reference for multi-turn agentic workflows with tool calling on the new gateway.","breadcrumbs":[{"label":"API Reference"}]},{"id":"EI0yGU3pIWgaco2nGlwT","title":"Embeddings","pathname":"/parasail-docs/api-reference/embeddings","siteSpaceId":"sitesp_Wgad2","description":"Embeddings API for generating vector representations of text using open-source models.","breadcrumbs":[{"label":"API Reference"}]},{"id":"pcC3UoRE6s5a5P0r7q6i","title":"Batch API","pathname":"/parasail-docs/api-reference/batch-api","siteSpaceId":"sitesp_Wgad2","description":"OpenAI-compatible Batch API reference for Parasail batch processing.","breadcrumbs":[{"label":"API Reference"}]},{"id":"mPJilLno4YP2FG0rbGDb","title":"Billing API","pathname":"/parasail-docs/api-reference/billing-api","siteSpaceId":"sitesp_Wgad2","description":"Programmatically retrieve invoices, real-time month-to-date spend, and hourly or daily usage breakdowns with the Parasail Billing API.","breadcrumbs":[{"label":"API Reference"}]},{"id":"WF9PbdCf4a3Y3AT4cRu6","title":"Models Endpoint","pathname":"/parasail-docs/api-reference/models-endpoint","siteSpaceId":"sitesp_Wgad2","description":"List available models on the Parasail platform using the /v1/models endpoint.","breadcrumbs":[{"label":"API Reference"}]},{"id":"isW5tFEn2dv7ZLQiyN1G","title":"Parameters","pathname":"/parasail-docs/api-reference/parameters","siteSpaceId":"sitesp_Wgad2","description":"Sampling parameters for the Parasail chat completions and text completions APIs.","breadcrumbs":[{"label":"API Reference"}]},{"id":"3ca5XLoJwu5IfzeJ3VQv","title":"Chat Completions","pathname":"/parasail-docs/guides/chat-completions","siteSpaceId":"sitesp_Wgad2","description":"Send chat messages to instruct-tuned models with Parasail's OpenAI-compatible Chat Completions API.","breadcrumbs":[{"label":"Guides"}]},{"id":"szBGEZulKYFykthUNPPk","title":"RAG","pathname":"/parasail-docs/guides/rag","siteSpaceId":"sitesp_Wgad2","description":"Build Retrieval-Augmented Generation systems that combine embeddings, vector retrieval, and LLM generation on Parasail.","breadcrumbs":[{"label":"Guides"}]},{"id":"CU73BGK8uWsIfRCDqYSP","title":"Multi-Modal","pathname":"/parasail-docs/guides/multi-modal","siteSpaceId":"sitesp_Wgad2","description":"Send images to vision-language models like Qwen2.5-VL using base64 data URLs through Parasail's OpenAI-compatible API.","breadcrumbs":[{"label":"Guides"}]},{"id":"cIbi74F17sjiSPonlTSz","title":"Structured Output","pathname":"/parasail-docs/guides/structured-output","siteSpaceId":"sitesp_Wgad2","description":"Get reliable, schema-conformant JSON from Parasail models using guided_json and response_format.","breadcrumbs":[{"label":"Guides"}]},{"id":"RDmqc5Syp8wTDrIHtPOi","title":"Tool/Function Calling","pathname":"/parasail-docs/guides/tool-function-calling","siteSpaceId":"sitesp_Wgad2","description":"Let Parasail models call your functions and tools via the Chat Completions and Responses APIs.","breadcrumbs":[{"label":"Guides"}]},{"id":"0YeMHiehtUWYWJBRhp0l","title":"Model Selection","pathname":"/parasail-docs/guides/model-recommendations","siteSpaceId":"sitesp_Wgad2","description":"Choose Parasail models by capability, deployment tier, latency, cost, and validation workflow instead of relying on stale static rankings.","breadcrumbs":[{"label":"Guides"}]},{"id":"8hnW6Po2Bj1dutmH0v2j","title":"Pricing","pathname":"/parasail-docs/billing/pricing","siteSpaceId":"sitesp_Wgad2","description":"Understand Parasail pricing across the Serverless per-token, Dedicated per-GPU-hour, and Batch discounted per-token tiers.","breadcrumbs":[{"label":"Billing"}]},{"id":"Vht8H5riRxpPrne4jhm0","title":"Overview","pathname":"/parasail-docs/operate-in-production/overview","siteSpaceId":"sitesp_Wgad2","description":"Run Parasail in production with rate-limit handling, quota planning, dedicated scaling, batch operations, and safe retries.","breadcrumbs":[{"label":"Operate in Production"}]},{"id":"brV3904cWnPfgDwsLKER","title":"Retries and Idempotency","pathname":"/parasail-docs/operate-in-production/retries-and-idempotency","siteSpaceId":"sitesp_Wgad2","description":"Handle 429 responses and transient errors with exponential backoff, sensible timeouts, and safe retries against the Parasail API.","breadcrumbs":[{"label":"Operate in Production"}]},{"id":"lbfcmNhaypsFOBdiIX7B","title":"Limits and Quotas","pathname":"/parasail-docs/operate-in-production/limits-and-quotas","siteSpaceId":"sitesp_Wgad2","description":"Rate limits, GPU quotas, and quota increase guidance for Parasail production workloads.","breadcrumbs":[{"label":"Operate in Production"}]},{"id":"SQlkm6H1FBnmbAoEaJ2Z","title":"Security Overview","pathname":"/parasail-docs/security-and-account-management/overview","siteSpaceId":"sitesp_Wgad2","description":"A security and account-management entry point for Parasail data handling, privacy, compliance, and API-key controls.","breadcrumbs":[{"label":"Security and Account Management"}]},{"id":"v2zkpJl60FfHuBqWAC6A","title":"Account and API Keys","pathname":"/parasail-docs/security-and-account-management/account-api-keys","siteSpaceId":"sitesp_Wgad2","description":"Manage Parasail organizations, account access, and read-only API keys.","breadcrumbs":[{"label":"Security and Account Management"}]},{"id":"9NzGG4P8mD9sCd9NCJMK","title":"Chat and Text Generation","pathname":"/parasail-docs/use-cases/chat-text-generation","siteSpaceId":"sitesp_Wgad2","description":"Build chatbots, assistants, and text generation pipelines with Parasail's OpenAI-compatible API.","breadcrumbs":[{"label":"Use Cases"}]},{"id":"VgTsKmx1uUyZtUKwy1WK","title":"RAG and Embeddings","pathname":"/parasail-docs/use-cases/rag-embeddings","siteSpaceId":"sitesp_Wgad2","description":"Build retrieval-augmented generation (RAG) pipelines and vector search systems with Parasail embeddings.","breadcrumbs":[{"label":"Use Cases"}]},{"id":"AXpJHRJH9bgHtkexfQ2q","title":"Batch Processing at Scale","pathname":"/parasail-docs/use-cases/batch-processing","siteSpaceId":"sitesp_Wgad2","description":"Process large volumes of LLM inferences, embeddings, and multimodal inputs at scale with Parasail Batch.","breadcrumbs":[{"label":"Use Cases"}]},{"id":"AwBBAik6tiXQRBQzxem9","title":"Agents and Tool Calling","pathname":"/parasail-docs/use-cases/agents-tool-calling","siteSpaceId":"sitesp_Wgad2","description":"Build agentic workflows with function calling, multi-step reasoning, and tool use on Parasail.","breadcrumbs":[{"label":"Use Cases"}]}]}