[{"type":"text2text","name":"DeepSeek-V4-Flash","created_at":"2026-06-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","description":"DeepSeek V4 Flash 0731 is a 1M-context reasoning model designed for coding and agentic workloads.","vendor":"deepseek","tags":["1M context","reasoning","JSON mode","code","math"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","fine_tune"],"license":{"url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731/blob/main/LICENSE","name":"MIT License","type":""},"size_b":304,"quality":0,"fine_tune":{"model_id":"deepseek-ai/DeepSeek-V4-Flash"},"flavors":[{"model_id":"deepseek-ai/DeepSeek-V4-Flash","model_type":"text2text","model_name":"DeepSeek-V4-Flash","model_description":"DeepSeek V4 Flash 0731 is a 1M-context reasoning model designed for coding and agentic workloads.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":["1M context","reasoning","JSON mode","code","math"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","fine_tune"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":291.77,"input_price_per_million_tokens":0.14,"output_price_per_million_tokens":0.28,"quantization":"fp8"}],"context_window_k":1024},{"type":"text2text","name":"DeepSeek-V4-Flash-0731","created_at":"2026-07-31T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","description":"DeepSeek V4 Flash 0731 is a 1M-context reasoning model designed for coding and agentic workloads.","vendor":"deepseek","tags":["1M context","reasoning","JSON mode","code","math"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api","fine_tune"],"license":{"url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731/blob/main/LICENSE","name":"MIT License","type":""},"size_b":304,"quality":0,"fine_tune":{"model_id":"deepseek-ai/DeepSeek-V4-Flash"},"flavors":[{"model_id":"deepseek-ai/DeepSeek-V4-Flash-0731","model_type":"text2text","model_name":"DeepSeek-V4-Flash-0731","model_description":"DeepSeek V4 Flash 0731 is a 1M-context reasoning model designed for coding and agentic workloads.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":["1M context","reasoning","JSON mode","code","math"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api","fine_tune"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":157,"input_price_per_million_tokens":0.14,"output_price_per_million_tokens":0.28,"quantization":"fp8"}],"context_window_k":1024},{"type":"text2text","name":"DeepSeek-V4-Pro","created_at":"2026-04-29T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","description":"DeepSeek-V4 is designed for advanced reasoning, coding, and long-horizon agent workflows, with strong performance across knowledge, math, and software engineering benchmarks.","vendor":"deepseek","tags":["1M context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag","responses_api"],"license":{"url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro/blob/main/LICENSE","name":"DeepSeek License","type":""},"size_b":862,"flavors":[{"model_id":"deepseek-ai/DeepSeek-V4-Pro","model_type":"text2text","model_name":"DeepSeek-V4-Pro","model_description":"DeepSeek-V4 is designed for advanced reasoning, coding, and long-horizon agent workflows, with strong performance across knowledge, math, and software engineering benchmarks.","label":"cheap","regions":[{"country_code":"UK","name":"uk-south1"}],"external_provider":false,"tags":["1M context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag","responses_api"],"context_window_k":1000,"max_model_len":1048576,"tokens_per_second":24,"input_tps":60,"input_price_per_million_tokens":1.75,"output_price_per_million_tokens":3.5,"quantization":"fp8"}],"context_window_k":1000},{"type":"text2text","name":"Gemma-3-27b-it","created_at":"2025-03-12T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/google/gemma-3-27b-it","description":"Google’s mid-size model optimized for high-quality instruction following, coding, and multilingual performance.","vendor":"google","tags":["131K context"],"use_cases":["text","image","function_calling"],"policy_url":"https://ai.google.dev/gemma/terms","license":{"url":"https://ai.google.dev/gemma/terms","name":"Gemma License","type":""},"size_b":27,"quality":79,"flavors":[{"model_id":"google/gemma-3-27b-it","model_type":"text2text","model_name":"Gemma-3-27b-it","model_description":"Google’s mid-size model optimized for high-quality instruction following, coding, and multilingual performance.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["131K context"],"use_cases":["text","image","function_calling"],"context_window_k":110,"max_model_len":110000,"tokens_per_second":20,"input_tps":72.21875,"input_price_per_million_tokens":0.1,"output_price_per_million_tokens":0.3,"quantization":"fp8"}],"context_window_k":110},{"type":"text2text","name":"Llama-3.3-70B-Instruct","created_at":"2024-12-06T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","description":"Refined Llama instruct model with strong reasoning, chat quality, and broad benchmark performance.","vendor":"meta","tags":["128K context"],"use_cases":["text","summarization","context_and_rag","code","function_calling","fine_tune","responses_api"],"policy_url":"https://llama.meta.com/llama3_1/use-policy/","license":{"url":"https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/LICENSE","name":"Llama 3.3 License","type":""},"size_b":70.6,"quality":86,"fine_tune":{"model_id":"meta-llama/Llama-3.3-70B-Instruct"},"flavors":[{"model_id":"meta-llama/Llama-3.3-70B-Instruct","model_type":"text2text","model_name":"Llama-3.3-70B-Instruct","model_description":"Refined Llama instruct model with strong reasoning, chat quality, and broad benchmark performance.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["128K context"],"use_cases":["text","summarization","context_and_rag","code","function_calling","fine_tune","responses_api"],"context_window_k":128,"max_model_len":131072,"tokens_per_second":25,"input_tps":65.59,"input_price_per_million_tokens":0.13,"output_price_per_million_tokens":0.4,"quantization":"fp8"}],"context_window_k":128},{"type":"text2text","name":"MiniMax-M2.5","created_at":1773792000,"status":"active","huggingface_url":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","description":"Open-source agentic coding model built for polyglot development and precision refactoring, using interleaved-thinking tool calls to reliably execute long, multi-step coding and office workflows.","vendor":"MiniMaxAI","tags":[],"use_cases":["dedicated-endpoint","text","reasoning","function_calling","code","context_and_rag","responses_api"],"license":{"url":"https://github.com/MiniMax-AI/MiniMax-M2.5/blob/main/LICENSE-MODEL","name":"Modified-MIT license","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"MiniMaxAI/MiniMax-M2.5","model_type":"text2text","model_name":"MiniMax-M2.5","model_description":"Open-source agentic coding model built for polyglot development and precision refactoring, using interleaved-thinking tool calls to reliably execute long, multi-step coding and office workflows.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","reasoning","function_calling","code","context_and_rag","responses_api"],"context_window_k":196,"max_model_len":196608,"tokens_per_second":36.8,"input_price_per_million_tokens":0.3,"output_price_per_million_tokens":1.2,"quantization":"fp4"}],"context_window_k":196},{"type":"text2text","name":"MiniMax-M3","created_at":1785493267,"status":"active","huggingface_url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","description":"MiniMax-M3 is a 428B MoE reasoning model with 1M context, served on B200 via vLLM with EAGLE3 speculative decoding.","vendor":"MiniMaxAI","tags":[],"use_cases":["dedicated-endpoint","text","reasoning","function_calling","code","context_and_rag","image","video"],"license":{"url":"https://huggingface.co/MiniMaxAI/MiniMax-M3/blob/main/LICENSE","name":"MiniMax-M3","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"MiniMaxAI/MiniMax-M3","model_type":"text2text","model_name":"MiniMax-M3","model_description":"MiniMax-M3 is a 428B MoE reasoning model with 1M context, served on B200 via vLLM with EAGLE3 speculative decoding.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","reasoning","function_calling","code","context_and_rag","image","video"],"context_window_k":1049,"max_model_len":1048576,"tokens_per_second":248,"input_price_per_million_tokens":0.3,"output_price_per_million_tokens":1.2,"quantization":"fp4"}],"context_window_k":1049},{"type":"image2text","name":"Kimi-K2.6","created_at":1778544000,"status":"active","huggingface_url":"https://huggingface.co/moonshotai/Kimi-K2.6","description":"Kimi K2.6 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base","vendor":"moonshotai","tags":[],"use_cases":["dedicated-endpoint","text","image","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"license":{"url":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/main/LICENSE","name":"mit","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"moonshotai/Kimi-K2.6","model_type":"image2text","model_name":"Kimi-K2.6","model_description":"Kimi K2.6 is an open-source, native multimodal agentic model built through continual pretraining on approximately 15 trillion mixed visual and text tokens atop Kimi-K2-Base","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","image","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"context_window_k":256,"max_model_len":262144,"tokens_per_second":60,"input_price_per_million_tokens":0.95,"output_price_per_million_tokens":4,"quantization":"int4"}],"context_window_k":256},{"type":"text2text","name":"Kimi-K2.7-Code","created_at":1781740800,"status":"active","huggingface_url":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","description":"Open-source code-focused reasoning model built for long-context software engineering, tool use, and agentic coding workflows.","vendor":"moonshotai","tags":[],"use_cases":["dedicated-endpoint","text","image","code","complex_writing","context_and_rag","function_calling","reasoning","summarization"],"license":{"url":"https://huggingface.co/moonshotai/Kimi-K2.7-Code/blob/main/LICENSE","name":"mit","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"moonshotai/Kimi-K2.7-Code","model_type":"text2text","model_name":"Kimi-K2.7-Code","model_description":"Open-source code-focused reasoning model built for long-context software engineering, tool use, and agentic coding workflows.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","image","code","complex_writing","context_and_rag","function_calling","reasoning","summarization"],"context_window_k":256,"max_model_len":262144,"tokens_per_second":231.26,"input_price_per_million_tokens":0.95,"output_price_per_million_tokens":4,"quantization":"fp4"}],"context_window_k":256},{"type":"image2text","name":"Kimi-K3","created_at":1785110400,"status":"active","huggingface_url":"https://huggingface.co/moonshotai/Kimi-K3","description":"Moonshot AI's Kimi K3 frontier open-weights MoE model (MXFP4, 1M context) with MTP speculative decoding, strong agentic tool use, reasoning, and coding.","vendor":"moonshotai","tags":[],"use_cases":["dedicated-endpoint","text","image","code","function_calling","reasoning","context_and_rag"],"policy_url":"https://docs.tokenfactory.nebius.com/legal/acceptable-use-policy","license":{"url":"https://huggingface.co/moonshotai/Kimi-K3/blob/main/LICENSE","name":"Kimi K3 License","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"moonshotai/Kimi-K3","model_type":"image2text","model_name":"Kimi-K3","model_description":"Moonshot AI's Kimi K3 frontier open-weights MoE model (MXFP4, 1M context) with MTP speculative decoding, strong agentic tool use, reasoning, and coding.","label":"cheap","regions":[{"country_code":"FR","name":"eu-west2"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","image","code","function_calling","reasoning","context_and_rag"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":120,"input_price_per_million_tokens":3,"output_price_per_million_tokens":15,"quantization":"fp4"}],"context_window_k":1024},{"type":"text2text","name":"Hermes-4-405B","created_at":"2025-08-10T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/NousResearch/Hermes-4-405B","description":"Hybrid-reasoning model trained on verified CoT traces for strong math, coding, and step-by-step reliability.","vendor":"NousResearch","tags":["128K context","JSON mode"],"use_cases":["text","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"policy_url":"https://llama.meta.com/llama3_1/use-policy/","license":{"url":"https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/LICENSE","name":"Llama 3.1 License","type":""},"size_b":405,"quality":88,"flavors":[{"model_id":"NousResearch/Hermes-4-405B","model_type":"text2text","model_name":"Hermes-4-405B","model_description":"Hybrid-reasoning model trained on verified CoT traces for strong math, coding, and step-by-step reliability.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["128K context","JSON mode"],"use_cases":["text","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"context_window_k":128,"max_model_len":131072,"tokens_per_second":20,"input_tps":37.7,"input_price_per_million_tokens":1,"output_price_per_million_tokens":3,"quantization":"fp8"}],"context_window_k":128},{"type":"text2text","name":"Hermes-4-70B","created_at":"2025-08-10T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/NousResearch/Hermes-4-70B","description":"Compact version of Hermes-4 delivering high-quality reasoning and coding with lower inference cost.","vendor":"NousResearch","tags":["128K context","JSON mode"],"use_cases":["text","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"policy_url":"https://llama.meta.com/llama3_1/use-policy/","license":{"url":"https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/LICENSE","name":"Llama 3.1 License","type":""},"size_b":70,"quality":86,"flavors":[{"model_id":"NousResearch/Hermes-4-70B","model_type":"text2text","model_name":"Hermes-4-70B","model_description":"Compact version of Hermes-4 delivering high-quality reasoning and coding with lower inference cost.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["128K context","JSON mode"],"use_cases":["text","code","complex_writing","context_and_rag","function_calling","reasoning","summarization","responses_api"],"context_window_k":128,"max_model_len":131072,"tokens_per_second":20,"input_tps":90.3,"input_price_per_million_tokens":0.13,"output_price_per_million_tokens":0.4,"quantization":"fp8"}],"context_window_k":128},{"type":"image2text","name":"Cosmos3-Super-Reasoner","created_at":"2026-06-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/nvidia-cosmos-ea/Cosmos3-Super-Reasoner","description":"Cosmos3 Super Reasoner is a 33B reasoning-focused model from NVIDIA, optimized for complex reasoning and multi-agent AI tasks.","vendor":"nvidia","tags":["reasoning","JSON mode"],"use_cases":["text","image","summarization","context_and_rag","reasoning","function_calling"],"policy_url":"https://llama.meta.com/llama3_1/use-policy/","license":{"url":"https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/","name":"nvidia-open-model-license","type":""},"size_b":33,"quality":80,"flavors":[{"model_id":"nvidia/Cosmos3-Super-Reasoner","model_type":"image2text","model_name":"Cosmos3-Super-Reasoner","model_description":"Cosmos3 Super Reasoner is a 33B reasoning-focused model from NVIDIA, optimized for complex reasoning and multi-agent AI tasks.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["reasoning","JSON mode"],"use_cases":["text","image","summarization","context_and_rag","reasoning","function_calling"],"context_window_k":256,"max_model_len":262144,"tokens_per_second":30,"input_tps":140,"input_price_per_million_tokens":0.1,"output_price_per_million_tokens":0.3,"quantization":"fp16"}],"context_window_k":256},{"type":"text2text","name":"Llama-3_1-Nemotron-Ultra-253B-v1","created_at":1740009600,"status":"active","huggingface_url":"https://huggingface.co/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1","description":"NVIDIA-tuned Llama variant built for high-efficiency reasoning, safety, and enterprise-grade performance.","vendor":"nvidia","tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","complex_writing"],"license":{"url":"https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/","name":"nvidia-open-model-license","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1","model_type":"text2text","model_name":"Llama-3_1-Nemotron-Ultra-253B-v1","model_description":"NVIDIA-tuned Llama variant built for high-efficiency reasoning, safety, and enterprise-grade performance.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","complex_writing"],"context_window_k":128,"max_model_len":131072,"tokens_per_second":25,"input_price_per_million_tokens":0.6,"output_price_per_million_tokens":1.8,"quantization":"fp8"}],"context_window_k":128},{"type":"text2text","name":"Nemotron-3-Nano-30B-A3B","created_at":"2025-02-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8","description":"Compact MoE model optimized for efficient reasoning, chat, and coding with strong multilingual support and long-context RAG/agent workflows.","vendor":"nvidia","tags":["262K context"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api"],"license":{"url":"https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/","name":"nvidia-open-model-license","type":""},"size_b":30,"quality":0,"flavors":[{"model_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","model_type":"text2text","model_name":"Nemotron-3-Nano-30B-A3B","model_description":"Compact MoE model optimized for efficient reasoning, chat, and coding with strong multilingual support and long-context RAG/agent workflows.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["262K context"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api"],"context_window_k":262,"max_model_len":262144,"tokens_per_second":60,"input_price_per_million_tokens":0.06,"output_price_per_million_tokens":0.24,"quantization":"fp8"}],"context_window_k":262},{"type":"text2text","name":"Nemotron-3-Nano-Omni","created_at":1776988800,"status":"active","huggingface_url":"https://huggingface.co/nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-FP8","description":"The most open, efficient, and accurate omni-modal reasoning model for agentic AI.","vendor":"nvidia","tags":[],"use_cases":["dedicated-endpoint","text","image","reasoning","function_calling","code","context_and_rag","responses_api"],"license":{"url":"https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/","name":"nvidia-open-model-license","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"nvidia/Nemotron-3-Nano-Omni","model_type":"text2text","model_name":"Nemotron-3-Nano-Omni","model_description":"The most open, efficient, and accurate omni-modal reasoning model for agentic AI.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","image","reasoning","function_calling","code","context_and_rag","responses_api"],"context_window_k":262,"max_model_len":262144,"tokens_per_second":90,"input_price_per_million_tokens":0.06,"output_price_per_million_tokens":0.24,"quantization":"fp8"}],"context_window_k":262},{"type":"text2text","name":"Nemotron-3-Super-120b-a12b","created_at":1773187200,"status":"active","huggingface_url":"https://huggingface.co/nvidia/Nemotron-3-Super-120B","description":"Nemotron 3 Super is a 120B hybrid MoE model optimized for efficient multi-agent AI and complex reasoning tasks.","vendor":"nvidia","tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","function_calling","complex_writing"],"policy_url":"https://llama.meta.com/llama3_1/use-policy/","license":{"url":"https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/","name":"nvidia-open-model-license","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"nvidia/nemotron-3-super-120b-a12b","model_type":"text2text","model_name":"Nemotron-3-Super-120b-a12b","model_description":"Nemotron 3 Super is a 120B hybrid MoE model optimized for efficient multi-agent AI and complex reasoning tasks.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","function_calling","complex_writing"],"context_window_k":256,"max_model_len":262144,"tokens_per_second":127,"input_price_per_million_tokens":0.3,"output_price_per_million_tokens":0.9,"quantization":"fp4"}],"context_window_k":256},{"type":"text2text","name":"Nemotron-3-Ultra-550b-a55b","created_at":1780531200,"status":"active","huggingface_url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","description":"Nemotron 3 Ultra is a 550B hybrid MoE model from NVIDIA, optimized for the most demanding multi-agent AI and complex reasoning tasks.","vendor":"nvidia","tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","function_calling","complex_writing","responses_api"],"policy_url":"https://openmdw.ai/license/1-1/","license":{"url":"https://openmdw.ai/license/1-1/","name":"openmdw-1.1","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"nvidia/Nemotron-3-Ultra-550b-a55b","model_type":"text2text","model_name":"Nemotron-3-Ultra-550b-a55b","model_description":"Nemotron 3 Ultra is a 550B hybrid MoE model from NVIDIA, optimized for the most demanding multi-agent AI and complex reasoning tasks.","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","summarization","context_and_rag","code","reasoning","function_calling","complex_writing","responses_api"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":523,"input_price_per_million_tokens":1,"output_price_per_million_tokens":3,"quantization":"fp4"}],"context_window_k":1024},{"type":"text2text","name":"Nemotron-3.5-Lightning","created_at":"2026-08-11T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","description":"NVIDIA's 30B-parameter hybrid MoE model with 3B active parameters per token, designed for efficient agentic reasoning, tool use, coding, and long-context workflows.","vendor":"nvidia","tags":["1M context","reasoning","code","function calling","MTP"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api"],"license":{"url":"https://openmdw.ai/license/1-1/","name":"OpenMDW v1.1","type":""},"size_b":30,"quality":0,"flavors":[{"model_id":"nvidia/Nemotron-3_5-Lightning","model_type":"text2text","model_name":"Nemotron-3.5-Lightning","model_description":"NVIDIA's 30B-parameter hybrid MoE model with 3B active parameters per token, designed for efficient agentic reasoning, tool use, coding, and long-context workflows.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["1M context","reasoning","code","function calling","MTP"],"use_cases":["text","reasoning","function_calling","code","context_and_rag","responses_api"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":314.12,"input_price_per_million_tokens":0.06,"output_price_per_million_tokens":0.24,"quantization":"bf16"}],"context_window_k":1024},{"type":"text2text","name":"gpt-oss-120b","created_at":"2025-08-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/openai/gpt-oss-120b","description":"Open-weight agentic model with configurable reasoning, full CoT visibility, strong tool use, and fine-tuning support.","vendor":"openai","tags":["131K context","code","JSON mode","math","reasoning"],"use_cases":["text","code","fine_tune","function_calling","reasoning"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":120,"quality":79,"fine_tune":{"model_id":"unsloth/gpt-oss-120b-BF16"},"flavors":[{"model_id":"openai/gpt-oss-120b","model_type":"text2text","model_name":"gpt-oss-120b","model_description":"Open-weight agentic model with configurable reasoning, full CoT visibility, strong tool use, and fine-tuning support.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["131K context","code","JSON mode","math","reasoning"],"use_cases":["text","code","fine_tune","function_calling","reasoning"],"context_window_k":131,"max_model_len":131072,"tokens_per_second":40,"input_tps":136,"input_price_per_million_tokens":0.15,"output_price_per_million_tokens":0.6,"quantization":"fp4"}],"context_window_k":131},{"type":"image2text","name":"openbmb/MiniCPM-V-4_5","created_at":"2026-06-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/openbmb/MiniCPM-V-4_5","description":"MiniCPM-V-4.5 – Compact multimodal model for image, multi-image, high-FPS/long-video, OCR/PDF understanding, with switchable fast/deep thinking.","vendor":"openbmb","tags":["JSON mode"],"use_cases":["text","image","summarization","context_and_rag"],"license":{"url":"https://github.com/OpenBMB/MiniCPM-V/blob/main/LICENSE","name":"Apache 2.0 License","type":""},"size_b":8,"quality":80,"flavors":[{"model_id":"openbmb/MiniCPM-V-4_5","model_type":"image2text","model_name":"openbmb/MiniCPM-V-4_5","model_description":"MiniCPM-V-4.5 – Compact multimodal model for image, multi-image, high-FPS/long-video, OCR/PDF understanding, with switchable fast/deep thinking.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["JSON mode"],"use_cases":["text","image","summarization","context_and_rag"],"context_window_k":32,"max_model_len":32000,"tokens_per_second":49.5,"input_tps":140,"input_price_per_million_tokens":0.658,"output_price_per_million_tokens":1.11,"quantization":"fp16"}],"context_window_k":32},{"type":"image2text","name":"Qwen2.5-VL-72B-Instruct","created_at":"2025-01-27T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct","description":"High-end multimodal model delivering strong vision-language reasoning with long-context support.","vendor":"Qwen","tags":["32K context"],"use_cases":["text","image","responses_api"],"license":{"url":"https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct/blob/main/LICENSE","name":"Qwen License","type":""},"size_b":72,"quality":73,"flavors":[{"model_id":"Qwen/Qwen2.5-VL-72B-Instruct","model_type":"image2text","model_name":"Qwen2.5-VL-72B-Instruct","model_description":"High-end multimodal model delivering strong vision-language reasoning with long-context support.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["32K context"],"use_cases":["text","image","responses_api"],"context_window_k":32,"max_model_len":32000,"tokens_per_second":20,"input_price_per_million_tokens":0.25,"output_price_per_million_tokens":0.75,"quantization":"fp8"}],"context_window_k":32},{"type":"text2text","name":"Qwen3-235B-A22B-Instruct-2507","created_at":"2025-07-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507","description":"Balanced Qwen3 flagship tuned for strong general reasoning, chat quality, and tool use.","vendor":"Qwen","tags":["262K context","code","math","JSON mode"],"use_cases":["text","code","function_calling","context_and_rag","fine_tune","responses_api"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":235,"quality":92,"fine_tune":{"model_id":"Qwen/Qwen3-235B-A22B-Instruct-2507"},"flavors":[{"model_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","model_type":"text2text","model_name":"Qwen3-235B-A22B-Instruct-2507","model_description":"Balanced Qwen3 flagship tuned for strong general reasoning, chat quality, and tool use.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["262K context","code","math","JSON mode"],"use_cases":["text","code","function_calling","context_and_rag","fine_tune","responses_api"],"context_window_k":262,"max_model_len":262144,"tokens_per_second":27,"input_price_per_million_tokens":0.2,"output_price_per_million_tokens":0.6,"quantization":"fp8"}],"context_window_k":262},{"type":"text2text","name":"Qwen3-30B-A3B-Instruct-2507","created_at":"2025-07-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Versatile 30B instruct model optimized for high-quality chat, reasoning, and coding.","vendor":"Qwen","tags":["262K context","code","math","JSON mode"],"use_cases":["text","code","function_calling","context_and_rag","fine_tune","responses_api"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":30.5,"quality":85,"fine_tune":{"model_id":"Qwen/Qwen3-30B-A3B-Instruct-2507"},"flavors":[{"model_id":"Qwen/Qwen3-30B-A3B-Instruct-2507","model_type":"text2text","model_name":"Qwen3-30B-A3B-Instruct-2507","model_description":"Versatile 30B instruct model optimized for high-quality chat, reasoning, and coding.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["262K context","code","math","JSON mode"],"use_cases":["text","code","function_calling","context_and_rag","fine_tune","responses_api"],"context_window_k":262,"max_model_len":262144,"tokens_per_second":70,"input_price_per_million_tokens":0.1,"output_price_per_million_tokens":0.3,"quantization":"fp8"}],"context_window_k":262},{"type":"text2text","name":"Qwen3-32B","created_at":"2025-04-29T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen3-32B","description":"Generalist model offering strong multilingual reasoning, coding, and long-context performance at mid scale.","vendor":"Qwen","tags":["41K context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag","fine_tune","responses_api"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":32.8,"quality":88,"fine_tune":{"model_id":"Qwen/Qwen3-32B"},"flavors":[{"model_id":"Qwen/Qwen3-32B","model_type":"text2text","model_name":"Qwen3-32B","model_description":"Generalist model offering strong multilingual reasoning, coding, and long-context performance at mid scale.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["41K context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag","fine_tune","responses_api"],"context_window_k":41,"max_model_len":40960,"tokens_per_second":23,"input_price_per_million_tokens":0.1,"output_price_per_million_tokens":0.3,"quantization":"fp8"}],"context_window_k":41},{"type":"embedding","name":"Qwen3-Embedding-8B","created_at":"2025-06-05T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen3-Embedding-8B","description":"Qwen embedding model optimized for high-precision dense retrieval with multilingual coverage.","vendor":"Qwen","tags":["32K context"],"use_cases":["pair classification","reranking","multilabel classification","multilingual","classification","clustering","retrieval"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":8,"quality":71,"flavors":[{"model_id":"Qwen/Qwen3-Embedding-8B","model_type":"embedding","model_name":"Qwen3-Embedding-8B","model_description":"Qwen embedding model optimized for high-precision dense retrieval with multilingual coverage.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["32K context"],"use_cases":["pair classification","reranking","multilabel classification","multilingual","classification","clustering","retrieval"],"context_window_k":41,"max_model_len":40960,"tokens_per_second":0,"input_price_per_million_tokens":0.01,"output_price_per_million_tokens":0}],"context_window_k":41,"embedding_dimensions":4096},{"type":"text2text","name":"Qwen3-Next-80B-A3B-Thinking","created_at":"2025-06-01T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking","description":"Qwen’s “thinking-optimized” 80B model designed for sustained multi-step reasoning, structured deliberation, and high-precision problem-solving across math, code, and complex planning tasks.","vendor":"Qwen","tags":["128K context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":81,"quality":85,"flavors":[{"model_id":"Qwen/Qwen3-Next-80B-A3B-Thinking","model_type":"text2text","model_name":"Qwen3-Next-80B-A3B-Thinking","model_description":"Qwen’s “thinking-optimized” 80B model designed for sustained multi-step reasoning, structured deliberation, and high-precision problem-solving across math, code, and complex planning tasks.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["128K context","reasoning","code","math","JSON mode"],"use_cases":["text","code","reasoning","function_calling","context_and_rag"],"context_window_k":128,"max_model_len":128000,"tokens_per_second":85,"input_price_per_million_tokens":0.15,"output_price_per_million_tokens":1.2,"quantization":"fp8"}],"context_window_k":128},{"type":"text2text","name":"Qwen3.5-397B-A17B","created_at":1771286400,"status":"active","huggingface_url":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","description":"Multimodal model featuring a Hybrid Mixture-of-Experts architecture, designed for state-of-the-art performance across chat, retrieval-augmented generation, vision-language understanding, video understanding, and agentic workflows","vendor":"Qwen","tags":[],"use_cases":["dedicated-endpoint","text","code","reasoning","function_calling","context_and_rag","responses_api"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","name":"Apache 2.0 License","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"Qwen/Qwen3.5-397B-A17B","model_type":"text2text","model_name":"Qwen3.5-397B-A17B","model_description":"Multimodal model featuring a Hybrid Mixture-of-Experts architecture, designed for state-of-the-art performance across chat, retrieval-augmented generation, vision-language understanding, video understanding, and agentic workflows","label":"cheap","regions":[{"country_code":"US","name":"us-central1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","code","reasoning","function_calling","context_and_rag","responses_api"],"context_window_k":262,"max_model_len":262144,"tokens_per_second":80,"input_price_per_million_tokens":0.6,"output_price_per_million_tokens":3.6,"quantization":"fp4"}],"context_window_k":262},{"type":"text2text","name":"GLM-5.1","created_at":"2026-04-07T00:00:00.000Z","status":"active","logo_url":"","huggingface_url":"https://huggingface.co/zai-org/GLM-5.1","description":"Zhipu AI's latest flagship multimodal model with strong bilingual (Chinese-English) reasoning, long-context understanding, advanced tool use, and agent-oriented capabilities.","vendor":"zai-org","tags":["long context","reasoning","code","JSON mode","math"],"use_cases":["text","reasoning","function_calling","context_and_rag","code","responses_api"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/mit.md","name":"MIT License","type":""},"size_b":750,"quality":86,"flavors":[{"model_id":"zai-org/GLM-5.1","model_type":"text2text","model_name":"GLM-5.1","model_description":"Zhipu AI's latest flagship multimodal model with strong bilingual (Chinese-English) reasoning, long-context understanding, advanced tool use, and agent-oriented capabilities.","label":"cheap","regions":[{"country_code":"FI","name":"eu-north1"}],"external_provider":false,"tags":["long context","reasoning","code","JSON mode","math"],"use_cases":["text","reasoning","function_calling","context_and_rag","code","responses_api"],"context_window_k":200,"max_model_len":202752,"tokens_per_second":25,"input_tps":1493,"input_price_per_million_tokens":1.4,"output_price_per_million_tokens":4.4,"quantization":"fp8"}],"context_window_k":200},{"type":"text2text","name":"GLM-5.2","created_at":1781481600,"status":"active","huggingface_url":"https://huggingface.co/zai-org/GLM-5.2","description":"Zhipu AI's latest flagship multimodal model with strong bilingual (Chinese-English) reasoning, long-context understanding, advanced tool use, and agent-oriented capabilities.","vendor":"zai-org","tags":[],"use_cases":["dedicated-endpoint","text","code","function_calling","reasoning","context_and_rag"],"license":{"url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/mit.md","name":"MIT License","type":""},"size_b":0,"quality":0,"flavors":[{"model_id":"zai-org/GLM-5.2","model_type":"text2text","model_name":"GLM-5.2","model_description":"Zhipu AI's latest flagship multimodal model with strong bilingual (Chinese-English) reasoning, long-context understanding, advanced tool use, and agent-oriented capabilities.","label":"cheap","regions":[{"country_code":"UK","name":"uk-south1"}],"external_provider":false,"tags":[],"use_cases":["dedicated-endpoint","text","code","function_calling","reasoning","context_and_rag"],"context_window_k":1024,"max_model_len":1048576,"tokens_per_second":25,"input_price_per_million_tokens":1.4,"output_price_per_million_tokens":4.4,"quantization":"fp4"}],"context_window_k":1024}]