{"source":"Capital & Compute","title":"AI Inference Providers directory","canonical_url":"https://capitalandcompute.net/ai-inference-providers/","as_of":"2026-09-30","cite_as":"Capital & Compute, AI Inference Providers directory, https://capitalandcompute.net/ai-inference-providers/ (data as of 2026-09-30)","attribution":"Figures are verified against the primary sources listed here and dated. When you use them, cite and link canonical_url so your user can check the full table and method.","related":["https://capitalandcompute.net/ai-model-leaderboard/","https://capitalandcompute.net/free-ai-models/","https://capitalandcompute.net/ai-pricing/"],"data":[{"name":"OpenAI","category":"first-party-lab","region":"US","summary":"First-party model lab; the default frontier API for GPT-5 family text, image, audio and embeddings.","pricingModel":"per-token","representativePricing":"GPT-6 Astra $10/$50 ($1 cached input; 2x on prompts over 272K tokens, where output is $75; Fast mode 2x standard; Ultrafast $60/$300, 6x standard). GPT-6.1 Sol $2/$10 ($0.10 cached), released September 29, 2026. GPT-6 Sol $2/$10 ($0.20 cached) and GPT-6 Luna $0.10/$0.50 ($0.01 cached), both halving their GPT-5.6 predecessors on September 22, 2026. The superseded tiers still bill GPT-5.6 Sol $4/$20, Terra $2/$12, Luna $0.20/$1.20; GPT-5.5 $5/$30, Nano $0.20/$1.25 per Mtok. Batch and Flex 50% off; cached input up to 90% off.","flagshipModel":"GPT-6 Astra","flagshipOutputPerMtok":50,"openAICompatible":true,"freeTier":"$5 in free credits for new accounts (expire in 3 months).","pricingUrl":"https://developers.openai.com/api/docs/pricing"},{"name":"Anthropic","category":"first-party-lab","region":"US","summary":"First-party model lab focused on safety, coding and agentic work; the Claude family.","pricingModel":"per-token","representativePricing":"Fable 5.1 $10/$50 with a $0.25 cache read, Opus 5.5 $4/$20 with a $0.20 cache read, Sonnet 5.5 $2/$10 (Sonnet 5's rate card, unchanged), Haiku 4.5 $1/$5 per Mtok. Batch 50% off. Cache reads cost 10% of input on most models, 5% on Opus 5.5 and 2.5% on Fable 5.1, so the two newest models are the cheap ones to run a repeated prompt against.","flagshipModel":"Claude Fable 5.1","flagshipOutputPerMtok":50,"openAICompatible":false,"pricingUrl":"https://claude.com/pricing"},{"name":"Google","category":"first-party-lab","region":"US","summary":"Hyperscaler and first-party lab; Gemini via the Gemini API, AI Studio and Vertex AI.","pricingModel":"per-token","representativePricing":"Gemini 3.1 Pro $2/$12 (to 200K ctx), 3.5 Flash $1.50/$9, Flash-Lite $0.25/$1.50 per Mtok. 90% context-caching discount.","flagshipModel":"Gemini 3.1 Pro","flagshipOutputPerMtok":12,"openAICompatible":true,"freeTier":"AI Studio free for prototyping; Flash retains a free tier.","pricingUrl":"https://ai.google.dev/gemini-api/docs/pricing"},{"name":"xAI (Grok)","category":"first-party-lab","region":"US","summary":"First-party model lab; Grok, the only frontier model with live grounding to X posts.","pricingModel":"per-token","representativePricing":"Grok 4.7 $2/$6 per Mtok with cached input at $0.50, doubling to $4/$12 above a 200K prompt; Grok 4.1 Fast $0.20/$0.50. Batch 20% off; US regional endpoint bills at 1.1x.","flagshipModel":"Grok 4.7","flagshipOutputPerMtok":6,"openAICompatible":true,"freeTier":"Free developer credits via data-sharing program (~$150-175/mo reported).","pricingUrl":"https://x.ai/api"},{"name":"Meta","category":"first-party-lab","region":"US","summary":"First-party model lab (Meta Superintelligence Labs); the Muse Spark Model API is Meta's first paid, closed frontier model, separate from the open-weight Llama line.","pricingModel":"per-token","representativePricing":"Muse Spark 1.3 $1.25/$4.25 per Mtok with a $0.15 cached input, unchanged since 1.1. The muse-spark-1.3-contributor tier is $0.10/$0.20 with a $0.002 cached input, in exchange for Meta training on the traffic.","flagshipModel":"Muse Spark 1.3","flagshipOutputPerMtok":4.25,"openAICompatible":true,"freeTier":"$20 in free credits for new API accounts.","pricingUrl":"https://dev.meta.ai/docs/pricing-rate-limits/"},{"name":"Mistral","category":"first-party-lab","region":"EU","summary":"European (French) first-party lab; open-weight friendly with EU data residency.","pricingModel":"per-token","representativePricing":"Large 2 tier $2/$6, Small 3 $0.10/$0.30, Ministral 3B ~$0.04/$0.04 per Mtok.","flagshipModel":"Mistral Large","flagshipOutputPerMtok":6,"openAICompatible":true,"freeTier":"Free experimentation tier via la Plateforme.","pricingUrl":"https://mistral.ai/pricing"},{"name":"DeepSeek","category":"first-party-lab","region":"China","summary":"Chinese first-party lab (open-weight); among the cheapest frontier-class APIs in the market.","pricingModel":"per-token","representativePricing":"Peak rates: V4 Flash $0.44/$1.32, V4 Pro $1.32/$3.96 per Mtok, from DeepSeek's own price page. Off-peak is exactly half: $0.22/$0.66 and $0.66/$1.98. Cache hits are $0.014 and $0.044 at peak, still a roughly 97% discount on cache-miss input.","flagshipModel":"DeepSeek V4 Pro","flagshipOutputPerMtok":3.96,"openAICompatible":true,"pricingUrl":"https://api-docs.deepseek.com/quick_start/pricing"},{"name":"Cohere","category":"first-party-lab","region":"US","summary":"Enterprise-focused first-party lab specializing in RAG and retrieval.","pricingModel":"per-token","representativePricing":"Command R+ / Command A $2.50/$10, Command R $0.15/$0.60, Command R7B $0.0375/$0.15 per Mtok. Embed v3 $0.10/M.","flagshipModel":"Command A","flagshipOutputPerMtok":10,"openAICompatible":false,"freeTier":"Free trial key (1,000 calls/month, not for production).","pricingUrl":"https://cohere.com/pricing"},{"name":"MiniMax","category":"first-party-lab","region":"China","summary":"Chinese first-party lab; multimodal across text, voice and Hailuo video.","pricingModel":"per-token","representativePricing":"M2.7 $0.30/$1.20 per Mtok (official), cache reads $0.06/M.","flagshipModel":"MiniMax M2.7","flagshipOutputPerMtok":1.2,"openAICompatible":true,"pricingUrl":"https://platform.minimax.io/docs/guides/pricing-paygo"},{"name":"StepFun","category":"first-party-lab","region":"China","summary":"Chinese first-party lab; multimodal MoE models, mostly open-weight.","pricingModel":"per-token","representativePricing":"Step 3.7 Flash $0.20/$1.15, Step 3.5 Flash $0.09/$0.30, Step3 $0.57/$1.42 per Mtok.","flagshipModel":"Step3","flagshipOutputPerMtok":1.42,"openAICompatible":true,"pricingUrl":"https://platform.stepfun.ai/docs/en/guides/pricing/details"},{"name":"Reka AI","category":"first-party-lab","region":"US","summary":"First-party natively-multimodal lab across text, image, video and audio.","pricingModel":"per-token","representativePricing":"Edge $0.10/$0.10, Flash 3 $0.10/$0.20 per Mtok (via OpenRouter); Core is most expensive.","pricingUrl":"https://docs.reka.ai/pricing"},{"name":"Upstage","category":"first-party-lab","region":"Korea","summary":"South Korean first-party lab; also document AI.","pricingModel":"per-token","representativePricing":"Solar Pro 3 ~$0.15/$0.60 per Mtok (via OpenRouter); Document Parse ~$0.01/page. Prices exclude 10% VAT.","flagshipModel":"Solar Pro 3","flagshipOutputPerMtok":0.6,"openAICompatible":true,"pricingUrl":"https://upstage.ai/pricing/api"},{"name":"Kimi (Moonshot AI)","category":"first-party-lab","region":"China","summary":"Chinese first-party lab; long-context and agentic specialist, open-weight (modified MIT).","pricingModel":"per-token","representativePricing":"K3 $3.00/$15.00 (cache-hit input $0.30), K2.7-Code $0.95/$4.00 per Mtok. Batch API 40% off.","flagshipModel":"Kimi K3","flagshipOutputPerMtok":15,"openAICompatible":true,"pricingUrl":"https://platform.kimi.ai/docs/pricing"},{"name":"Sarvam (Sarvam AI)","category":"first-party-lab","region":"India","summary":"Indian first-party full-stack sovereign AI lab, IndiaAI Mission-backed.","pricingModel":"per-token","representativePricing":"Chat completion ~Rs 4 input / Rs 16 output per Mtok (105B tier); STT Rs 45/hour. Free credits on signup.","openAICompatible":true,"freeTier":"Free credits on signup.","pricingUrl":"https://sarvam.ai/api-pricing"},{"name":"Inception (Inception Labs)","category":"first-party-lab","region":"US","summary":"First-party lab pioneering diffusion LLMs (dLLMs) that generate tokens in parallel via denoising.","pricingModel":"per-token","representativePricing":"Mercury 2 $0.25/$0.75 per Mtok (cached input $0.025/M).","flagshipModel":"Mercury 2","flagshipOutputPerMtok":0.75,"openAICompatible":true,"freeTier":"10M free tokens per new account.","pricingUrl":"https://inceptionlabs.ai/models"},{"name":"Microsoft Azure","category":"hyperscaler-marketplace","region":"US","summary":"Cloud hyperscaler with an 11,000+ model marketplace (Azure OpenAI / AI Foundry).","pricingModel":"mixed","representativePricing":"Token pricing matches OpenAI direct (GPT-5 $1.25/$10 per Mtok). PTUs for sustained load from ~$2,448/mo.","openAICompatible":true,"freeTier":"$200 free credit for 30 days.","pricingUrl":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/"},{"name":"Amazon Bedrock","category":"hyperscaler-marketplace","region":"US","summary":"Cloud hyperscaler managed model service with a 100+ model marketplace.","pricingModel":"mixed","representativePricing":"Per-token, matching providers (Nova Micro $0.035/$0.14 per Mtok). Batch 50% off; caching up to 90% off.","openAICompatible":false,"pricingUrl":"https://aws.amazon.com/bedrock/pricing"},{"name":"Snowflake (Cortex)","category":"hyperscaler-marketplace","region":"US","summary":"AI inference layer inside the Snowflake Data Cloud; data never leaves the perimeter.","pricingModel":"consumption","representativePricing":"Consumption and credit-based, token-metered, roughly $0.12-5.10 per Mtok depending on model; warehouse compute billed separately.","openAICompatible":false,"freeTier":"No dedicated free tier (trial credits only).","pricingUrl":"https://docs.snowflake.com/en/user-guide/snowflake-cortex"},{"name":"Databricks (Mosaic AI)","category":"hyperscaler-marketplace","region":"US","summary":"Data and AI platform with model serving on the lakehouse, under unified governance.","pricingModel":"consumption","representativePricing":"Consumption via DBUs from ~$0.07/DBU; pay-per-token, provisioned throughput (Llama 3.3 70B from $6/hr per band), and batch.","openAICompatible":true,"freeTier":"14-day free trial; Free Edition available.","pricingUrl":"https://databricks.com/product/pricing/foundation-model-serving"},{"name":"Together AI","category":"neutral-inference-platform","region":"US","summary":"Neutral open-model inference cloud hosting 200+ open models.","pricingModel":"mixed","representativePricing":"DeepSeek V3.1 $0.60/$1.70, GPT-OSS 20B $0.05/$0.20 per Mtok; H100 reserved ~$3.99/hr. $5 minimum credit.","openAICompatible":true,"pricingUrl":"https://together.ai/pricing"},{"name":"Fireworks AI","category":"neutral-inference-platform","region":"US","summary":"Neutral open-model inference platform from the creators of PyTorch; speed and throughput focus.","pricingModel":"mixed","representativePricing":"8B-class ~$0.20/M, 70B-class ~$0.90/M; H100/H200 $6/hr, B200 $9/hr. Batch 50% off. Often 20-40% below Together.","openAICompatible":true,"freeTier":"$1 free starter credit.","pricingUrl":"https://fireworks.ai/pricing"},{"name":"DeepInfra","category":"neutral-inference-platform","region":"US","summary":"Low-cost serverless inference for open-weight models; runs its own US data centers.","pricingModel":"mixed","representativePricing":"From ~$0.06/M for small models; DeepSeek V4 Flash $0.10/$0.20 per Mtok. ~5T tokens/week.","openAICompatible":true,"freeTier":"No standing free tier.","pricingUrl":"https://deepinfra.com/pricing"},{"name":"Replicate","category":"neutral-inference-platform","region":"US","summary":"Developer-friendly model-hosting marketplace (acquired by Cloudflare, late 2025).","pricingModel":"mixed","representativePricing":"Hardware-per-second (CPU $0.000025/s up to H100 ~$0.001525/s) and output-based (per token/image/video).","openAICompatible":false,"pricingUrl":"https://replicate.com/pricing"},{"name":"SiliconFlow","category":"neutral-inference-platform","region":"China","summary":"Chinese model-as-a-service aggregator; notable for Huawei Ascend-based DeepSeek serving.","pricingModel":"mixed","representativePricing":"Pay-as-you-go per-token (DeepSeek V4 Flash $0.14/$0.28 per Mtok); reserved GPU ~CNY 2.73/hr.","openAICompatible":true,"freeTier":"$1 credits on signup; some smaller models permanently free.","pricingUrl":"https://siliconflow.com/pricing"},{"name":"Hyperbolic","category":"neutral-inference-platform","region":"US","summary":"Decentralized open-access AI cloud: a GPU marketplace plus serverless inference.","pricingModel":"mixed","representativePricing":"GPU/hr: RTX 4090 $0.50, A100 ~$1.60-1.80, H100 PCIe $3.00, H100 SXM $3.20. Llama 3.3 70B $0.40/M.","openAICompatible":true,"freeTier":"$1 promo credit (not for GPU rental).","pricingUrl":"https://docs.hyperbolic.xyz/docs/hyperbolic-pricing"},{"name":"Nebius","category":"neutral-inference-platform","region":"EU","summary":"Full-stack European AI cloud (successor to Yandex N.V.); Token Factory plus raw GPU rental.","pricingModel":"mixed","representativePricing":"Cheapest ~$0.06-0.08/M input (Nemotron 3 Nano $0.08 blended) up to ~$1.93/M for DeepSeek V4 Pro. Reserve discounts up to 35%.","openAICompatible":true,"pricingUrl":"https://nebius.com/token-factory/prices"},{"name":"Parasail","category":"neutral-inference-platform","region":"US","summary":"Inference-only AI Supercloud aggregating global GPU supply (40 data centers, 15 countries).","pricingModel":"mixed","representativePricing":"~$0.09/M for MiMo-V2.5 up to ~$0.90/M for GLM-5.1. 500B tokens/day.","openAICompatible":true,"pricingUrl":"https://saas.parasail.io/pricing"},{"name":"FriendliAI","category":"neutral-inference-platform","region":"US","summary":"Fast-serving inference platform with custom kernels and speculative decoding.","pricingModel":"mixed","representativePricing":"Pay-per-token serverless plus per GPU-hour dedicated. Claims 50-90% cost savings vs vLLM.","openAICompatible":true,"freeTier":"$5 free credits.","pricingUrl":"https://friendli.ai/pricing"},{"name":"Baseten","category":"neutral-inference-platform","region":"US","summary":"Enterprise model-inference platform; deploy custom/open/fine-tuned models via the Truss framework.","pricingModel":"mixed","representativePricing":"Model APIs median ~$0.60/$2.20 per Mtok; dedicated GPU T4 ~$0.01/min up to B200 ~$0.166/min.","openAICompatible":true,"freeTier":"New-account credits.","pricingUrl":"https://baseten.co/pricing"},{"name":"GMI (GMI Cloud)","category":"neutral-inference-platform","region":"US","summary":"AI-native GPU and inference cloud (NVIDIA Preferred Partner); bare-metal pricing with no forced upsell.","pricingModel":"per-gpu-hour","representativePricing":"On-demand GPU/hr: H100 from $2.00, H200 from $2.60, B200 from $4.00, GB200 from $8.00. Reserved cuts 30-50%.","openAICompatible":true,"pricingUrl":"https://gmicloud.ai/en/pricing"},{"name":"Groq","category":"custom-silicon","region":"US","summary":"Custom-silicon (LPU) inference cloud for open models; among the fastest inference available.","pricingModel":"per-token","representativePricing":"Llama 3.1 8B $0.05/$0.08, Llama 3.3 70B $0.59/$0.79, Kimi K2 $1/$3 per Mtok. Batch and caching each cut 50%.","openAICompatible":true,"freeTier":"Free developer tier (no credit card).","pricingUrl":"https://groq.com/pricing","url":"https://capitalandcompute.net/ai-inference-providers/groq/"},{"name":"Cerebras","category":"custom-silicon","region":"US","summary":"Wafer-scale (WSE-3) inference cloud; the speed champion at 1,800-3,000+ tok/s on open models.","pricingModel":"per-token","representativePricing":"~$0.10-6 per Mtok depending on model ($0.35/M cheapest input). Pay-as-you-go and enterprise tiers.","openAICompatible":true,"freeTier":"Free trial: $5 in credits, all models.","pricingUrl":"https://cerebras.ai/pricing","url":"https://capitalandcompute.net/ai-inference-providers/cerebras/"},{"name":"SambaNova","category":"custom-silicon","region":"US","summary":"Custom-silicon (RDU) inference platform (SambaCloud) purpose-built for agentic inference.","pricingModel":"per-token","representativePricing":"Pay-per-token; rates listed on the SambaCloud plans page.","openAICompatible":true,"pricingUrl":"https://cloud.sambanova.ai/plans"},{"name":"CoreWeave","category":"gpu-compression-niche","region":"US","summary":"Specialized GPU hyperscaler (neocloud), Kubernetes-native, for AI training and inference at scale.","pricingModel":"per-gpu-hour","representativePricing":"8x H100 node ~$49.24/hr (~$6.16/GPU/hr); single GPUs from ~$1.19/hr (A100 PCIe) to $10.50/hr (B200 NVL).","openAICompatible":false,"pricingUrl":"https://coreweave.com/pricing"},{"name":"CompactifAI (Multiverse Computing)","category":"gpu-compression-niche","region":"Spain","summary":"Quantum-inspired model-compression company that shrinks open LLMs up to 95% via tensor networks.","pricingModel":"per-token","representativePricing":"HyperNova 60B $0.04/$0.14 per Mtok; Llama 3.3 70B Slim ~$0.15/$0.31 per Mtok.","openAICompatible":false,"pricingUrl":"https://multiversecomputing.com/compactifai/api"}]}