{"ok":true,"source":"tensorfeed.ai","lastUpdated":"2026-09-21","tracked_providers":["Together AI","Fireworks","DeepInfra","Groq","OpenRouter","Replicate","Anyscale","DeepSeek","GitHub Models","Venice"],"count":9,"models":[{"modelId":"llama-4-maverick","modelName":"Llama 4 Maverick","family":"Meta","paramsB":400,"license":"Llama 4 Community License","openWeights":true,"offers":[{"provider":"DeepInfra","providerModelId":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","inputPrice":0.2,"outputPrice":0.8,"blendedPrice":0.5,"contextWindow":1000000,"outputTPS":110,"features":["function-calling","vision"],"url":"https://deepinfra.com/pricing","note":"FP8 checkpoint; the only per-token host of Maverick left outside OpenRouter"},{"provider":"OpenRouter","providerModelId":"meta-llama/llama-4-maverick","inputPrice":0.1875,"outputPrice":0.6525,"blendedPrice":0.42,"contextWindow":1000000,"outputTPS":null,"features":["function-calling","vision"],"url":"https://openrouter.ai/meta-llama/llama-4-maverick","note":"Routes across multiple providers automatically"}]},{"modelId":"llama-4-scout","modelName":"Llama 4 Scout","family":"Meta","paramsB":109,"license":"Llama 4 Community License","openWeights":true,"offers":[{"provider":"DeepInfra","providerModelId":"meta-llama/Llama-4-Scout-17B-16E-Instruct","inputPrice":0.1,"outputPrice":0.3,"blendedPrice":0.2,"contextWindow":10000000,"outputTPS":170,"features":["function-calling","vision"],"url":"https://deepinfra.com/pricing","note":""},{"provider":"OpenRouter","providerModelId":"meta-llama/llama-4-scout","inputPrice":0.1,"outputPrice":0.3,"blendedPrice":0.2,"contextWindow":10000000,"outputTPS":null,"features":["function-calling","vision"],"url":"https://openrouter.ai/meta-llama/llama-4-scout","note":"Same headline price as DeepInfra"}]},{"modelId":"llama-3.1-405b","modelName":"Llama 3.1 405B Instruct","family":"Meta","paramsB":405,"license":"Llama 3.1 Community License","openWeights":true,"offers":[{"provider":"OpenRouter","providerModelId":"meta-llama/llama-3.1-405b-instruct","inputPrice":3,"outputPrice":3,"blendedPrice":3,"contextWindow":130000,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/meta-llama/llama-3.1-405b-instruct","note":"Only per-token route left in this matrix; Together, Fireworks and DeepInfra dropped serverless 405B"}]},{"modelId":"llama-3.1-70b","modelName":"Llama 3.1 70B Instruct","family":"Meta","paramsB":70,"license":"Llama 3.1 Community License","openWeights":true,"offers":[{"provider":"OpenRouter","providerModelId":"meta-llama/llama-3.1-70b-instruct","inputPrice":0.4,"outputPrice":0.4,"blendedPrice":0.4,"contextWindow":130000,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/meta-llama/llama-3.1-70b-instruct","note":"Only per-token route left in this matrix; Groq now serves Llama 3.3 70B in its place"}]},{"modelId":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","family":"DeepSeek","paramsB":1600,"license":"MIT","openWeights":true,"offers":[{"provider":"Together AI","providerModelId":"deepseek-ai/DeepSeek-V4-Pro","inputPrice":1.32,"outputPrice":3.96,"blendedPrice":2.64,"contextWindow":1000000,"outputTPS":90,"features":["function-calling","json-mode"],"url":"https://www.together.ai/pricing","note":"Listed as DeepSeek V4 Pro 0813; cached input is $0.13"},{"provider":"Fireworks","providerModelId":"accounts/fireworks/models/deepseek-v4-pro","inputPrice":1.32,"outputPrice":3.96,"blendedPrice":2.64,"contextWindow":1000000,"outputTPS":85,"features":["function-calling","json-mode"],"url":"https://docs.fireworks.ai/serverless/pricing","note":"Per-token rate is published for the 0813 snapshot; the base model id is on-demand deployment only"},{"provider":"OpenRouter","providerModelId":"deepseek/deepseek-v4-pro","inputPrice":0.435,"outputPrice":0.87,"blendedPrice":0.6525,"contextWindow":1048576,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/deepseek/deepseek-v4-pro","note":"Cheapest paid route here by a wide margin. The old deepseek/deepseek-chat slug tracked an alias DeepSeek retired on 2026-07-24"},{"provider":"GitHub Models","providerModelId":"DeepSeek-V4-Pro","inputPrice":0,"outputPrice":0,"blendedPrice":0,"contextWindow":1000000,"outputTPS":null,"features":["function-calling"],"url":"https://github.com/marketplace/models","note":"Free tier rate-limited (prototyping); paid via Azure AI Foundry"},{"provider":"Venice","providerModelId":"deepseek-v4-pro","inputPrice":1.65,"outputPrice":3.3,"blendedPrice":2.475,"contextWindow":1000000,"outputTPS":null,"features":["function-calling","json-mode"],"url":"https://docs.venice.ai/overview/pricing","note":"Privacy-first host: no prompt retention. Venice prices the newer deepseek-v4-pro-0813 snapshot higher, at $1.65 in and $4.95 out"}]},{"modelId":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","family":"DeepSeek","paramsB":70,"license":"MIT","openWeights":true,"offers":[{"provider":"Together AI","providerModelId":"deepseek-ai/DeepSeek-V4-Flash","inputPrice":0.14,"outputPrice":0.28,"blendedPrice":0.21,"contextWindow":130000,"outputTPS":145,"features":["function-calling","json-mode"],"url":"https://www.together.ai/pricing","note":"Listed as DeepSeek V4 Flash 0731; cached input is $0.03"},{"provider":"Fireworks","providerModelId":"accounts/fireworks/models/deepseek-v4-flash","inputPrice":0.22,"outputPrice":0.66,"blendedPrice":0.44,"contextWindow":130000,"outputTPS":130,"features":["function-calling","json-mode"],"url":"https://docs.fireworks.ai/serverless/pricing","note":"Per-token rate is published for the 0731 snapshot"},{"provider":"OpenRouter","providerModelId":"deepseek/deepseek-v4-flash","inputPrice":0.05544,"outputPrice":0.1109,"blendedPrice":0.08317,"contextWindow":1000000,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/deepseek/deepseek-v4-flash/pricing","note":"Promotional rate, listed as 60 percent off standard, so it can move without notice"},{"provider":"Venice","providerModelId":"deepseek-v4-flash","inputPrice":0.14,"outputPrice":0.28,"blendedPrice":0.21,"contextWindow":1000000,"outputTPS":null,"features":["function-calling","json-mode"],"url":"https://docs.venice.ai/overview/pricing","note":"Serves the full 1M context where Together and Fireworks cap this model at 130k"}]},{"modelId":"mixtral-8x22b","modelName":"Mixtral 8x22B Instruct","family":"Mistral","paramsB":141,"license":"Apache-2.0","openWeights":true,"offers":[{"provider":"OpenRouter","providerModelId":"mistralai/mixtral-8x22b-instruct","inputPrice":0.65,"outputPrice":0.65,"blendedPrice":0.65,"contextWindow":65536,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/mistralai/mixtral-8x22b-instruct","note":"Only per-token route left in this matrix; DeepInfra removed the model and Fireworks moved it to dedicated deployment"}]},{"modelId":"phi-4","modelName":"Phi-4","family":"Microsoft","paramsB":14,"license":"MIT","openWeights":true,"offers":[{"provider":"GitHub Models","providerModelId":"Phi-4","inputPrice":0,"outputPrice":0,"blendedPrice":0,"contextWindow":16384,"outputTPS":null,"features":["function-calling"],"url":"https://github.com/marketplace/models/azureml/Phi-4","note":"Free tier rate-limited (prototyping); paid via Azure AI Foundry"},{"provider":"DeepInfra","providerModelId":"microsoft/phi-4","inputPrice":0.07,"outputPrice":0.14,"blendedPrice":0.105,"contextWindow":16384,"outputTPS":220,"features":["function-calling"],"url":"https://deepinfra.com/pricing","note":""},{"provider":"OpenRouter","providerModelId":"microsoft/phi-4","inputPrice":0.07,"outputPrice":0.14,"blendedPrice":0.105,"contextWindow":16384,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/microsoft/phi-4","note":""}]},{"modelId":"qwen-2.5-72b","modelName":"Qwen 2.5 72B Instruct","family":"Alibaba","paramsB":72,"license":"Qwen License","openWeights":true,"offers":[{"provider":"DeepInfra","providerModelId":"Qwen/Qwen2.5-72B-Instruct","inputPrice":0.36,"outputPrice":0.4,"blendedPrice":0.38,"contextWindow":130000,"outputTPS":80,"features":["function-calling"],"url":"https://deepinfra.com/pricing","note":""},{"provider":"OpenRouter","providerModelId":"qwen/qwen-2.5-72b-instruct","inputPrice":0.35,"outputPrice":0.4,"blendedPrice":0.375,"contextWindow":130000,"outputTPS":null,"features":["function-calling"],"url":"https://openrouter.ai/qwen/qwen-2.5-72b-instruct","note":""},{"provider":"GitHub Models","providerModelId":"qwen2.5-72b-instruct","inputPrice":0,"outputPrice":0,"blendedPrice":0,"contextWindow":130000,"outputTPS":null,"features":["function-calling"],"url":"https://github.com/marketplace/models","note":"Free tier rate-limited (prototyping); paid via Azure AI Foundry"}]}]}