{"source":"Capital & Compute","title":"Best Free AI Model APIs","canonical_url":"https://capitalandcompute.net/free-ai-models/","as_of":"2026-09-05","cite_as":"Capital & Compute, Best Free AI Model APIs, https://capitalandcompute.net/free-ai-models/ (data as of 2026-09-05)","attribution":"Figures are verified against the primary sources listed here and dated. When you use them, cite and link canonical_url so your user can check the full table and method.","related":["https://capitalandcompute.net/ai-inference-providers/","https://capitalandcompute.net/ai-model-leaderboard/","https://capitalandcompute.net/can-i-run-this-llm/"],"context":{"qualityGate":{"metric":"Artificial Analysis Intelligence Index","min":20,"note":"the index is demanding: on v4.2 the best model of any price scores 57 and the strongest free model here scores 53, so the 0-100 scale runs low. A free model scoring at least 20 is genuinely capable. The v4.2 recalibration pushed several previously listed free models below the bar, and they are no longer ranked: OpenAI gpt-oss-120b (16), NVIDIA Nemotron 3 Super (19), Google Gemma 4 26B A4B (19), NVIDIA Nemotron 3.5 Lightning (16) and Alibaba Qwen3 Next 80B A3B (8). They are still free; they are just no longer models we would tell you to build on."},"hosts":[{"key":"google-ai-studio","name":"Google AI Studio","freeSummary":"Gemini Flash and Flash-Lite free (Pro tiers left the free tier on 2026-04-01); roughly 5-15 req/min and 20-1,500 req/day depending on model","requiresSignup":true,"dataUsedForTraining":true,"sourceUrl":"https://ai.google.dev/gemini-api/docs/rate-limits"},{"key":"openrouter","name":"OpenRouter","freeSummary":"22 models tagged :free at 20 req/min and 50 req/day (1,000/day after a one-time $10 credit purchase)","requiresSignup":true,"dataUsedForTraining":false,"sourceUrl":"https://openrouter.ai/docs/api-reference/limits"},{"key":"opencode-zen","name":"OpenCode Zen","freeSummary":"31 models free through the OpenCode CLI and Desktop, including Muse Spark 1.3 and 1.2, MiMo V2.5, Nemotron 3 Ultra, MiniMax M3, DeepSeek V4 Flash and the Big Pickle stealth model","requiresSignup":true,"dataUsedForTraining":true,"sourceUrl":"https://models.dev/api.json"},{"key":"nvidia-nim","name":"NVIDIA NIM (build.nvidia.com)","freeSummary":"Open models free at about 40 req/min after phone verification","requiresSignup":true,"dataUsedForTraining":false,"sourceUrl":"https://build.nvidia.com/"},{"key":"groq","name":"Groq","freeSummary":"Open models free at roughly 1,000 req/day on larger models, up to 14,400 on small ones; 12K tokens/min","requiresSignup":true,"dataUsedForTraining":false,"sourceUrl":"https://console.groq.com/docs/rate-limits"},{"key":"cerebras","name":"Cerebras","freeSummary":"Open models free at 30 req/min, 14,400 req/day, 60K tokens/min","requiresSignup":true,"dataUsedForTraining":false,"sourceUrl":"https://inference-docs.cerebras.ai/support/pricing"},{"key":"cloudflare","name":"Cloudflare Workers AI","freeSummary":"10,000 neurons/day free across the model catalog (a usage credit, not a request cap)","requiresSignup":true,"dataUsedForTraining":false,"sourceUrl":"https://developers.cloudflare.com/workers-ai/platform/pricing/"}]},"data":[{"model":"Muse Spark 1.3","maker":"Meta","aaIndex":53,"contextLength":1048576,"openWeight":false,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Muse Spark 1.2","maker":"Meta","aaIndex":47,"contextLength":1048576,"openWeight":false,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"GLM-5.2","maker":"Z AI","aaIndex":43,"contextLength":256000,"openWeight":true,"hosts":[{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"}]},{"model":"DeepSeek V4 Flash 0731","maker":"DeepSeek","aaIndex":41,"contextLength":200000,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Gemini 3.5 Flash","maker":"Google","aaIndex":38,"contextLength":1048576,"openWeight":false,"hosts":[{"key":"google-ai-studio","name":"Google AI Studio","accessUrl":"https://aistudio.google.com/"}]},{"model":"MiniMax M3","maker":"MiniMax","aaIndex":36,"contextLength":1048576,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"},{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"}]},{"model":"Hy3","maker":"Tencent","aaIndex":33,"contextLength":190000,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Inkling","maker":"Thinking Machines","aaIndex":32,"contextLength":1048576,"openWeight":true,"hosts":[{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"}]},{"model":"Inkling Small","maker":"Thinking Machines","aaIndex":32,"contextLength":1048576,"openWeight":true,"hosts":[{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"}]},{"model":"Nemotron 3 Ultra","maker":"NVIDIA","aaIndex":30,"contextLength":1000000,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"},{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"},{"key":"nvidia-nim","name":"NVIDIA NIM (build.nvidia.com)","accessUrl":"https://build.nvidia.com/"}]},{"model":"MiMo V2.5","maker":"Xiaomi","aaIndex":30,"contextLength":200000,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"MiMo V2 Omni","maker":"Xiaomi","aaIndex":28,"contextLength":262144,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Ling 3.0 Flash","maker":"InclusionAI","aaIndex":27,"contextLength":262144,"openWeight":false,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"LongCat 2.0","maker":"LongCat","aaIndex":26,"contextLength":1000000,"openWeight":false,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"MiMo V2 Flash","maker":"Xiaomi","aaIndex":26,"contextLength":262144,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Ring 2.6 1T","maker":"InclusionAI","aaIndex":24,"contextLength":262000,"openWeight":true,"hosts":[{"key":"opencode-zen","name":"OpenCode Zen","accessUrl":"https://opencode.ai/docs/zen/"}]},{"model":"Gemma 4 31B","maker":"Google","aaIndex":22,"contextLength":262144,"openWeight":true,"hosts":[{"key":"openrouter","name":"OpenRouter","accessUrl":"https://openrouter.ai/models?max_price=0"}]}]}