{"data":[{"id":"moonshot-kimi-k3","name":"kimi-k3","display_name":"Kimi K3","description":"A flagship reasoning-capable multimodal LLM with 2.8 trillion parameters, built on Kimi Delta Attention (a hybrid linear attention mechanism) with native visual understanding and tool-use support.","creator":"moonshot","family":"kimi","tier":"","version":"k3","type":"language","size_in_bn":2779.932,"modalities":{"input":["image","pdf","text","video"],"output":["text"]},"context_window":1048576,"max_output_tokens":131072,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-07-16","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":4,"ids":["accounts/fireworks/models/kimi-k3","kimi-k3","kimi-k3-low","moonshot-kimi-k3","moonshotai/kimi-k3","moonshotai/Kimi-K3"],"hf_likes":7099,"hf_downloads":2850,"hf_downloads_all_time":2855,"hf_trending_score":6642,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"moonshot-kimi-k3","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":3,"max_input_per_1m":3,"min_output_per_1m":15,"max_output_per_1m":15,"min_cache_read_per_1m":0.3,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["huggingface","openrouter","vercel_ai_gateway"],"provider_count":3},"providers":[],"regions":[],"region_info":{}}},{"id":"deepseek-v4-pro","name":"deepseek-v4-pro","display_name":"DeepSeek V4 Pro","description":"A large-scale Mixture-of-Experts LLM from DeepSeek with 1.6T total and 49B activated parameters, supporting a 1M-token context window for advanced reasoning and tool-use tasks.","creator":"deepseek","family":"v4","tier":"pro","version":null,"type":"language","size_in_bn":861.608,"modalities":{"input":["image","pdf","text"],"output":["text"]},"context_window":1048600,"max_output_tokens":393216,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"DeepSeek","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":true,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-04-24","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":9,"ids":["accounts/fireworks/models/deepseek-v4-pro","azure_ai/deepseek-v4-pro","dashscope/deepseek-v4-pro","deepseek-ai/DeepSeek-V4-Pro","deepseek-v4-pro","deepseek-v4-pro-high","deepseek-v4-pro-non-reasoning","deepseek/deepseek-v4-pro","fireworks_ai/accounts/fireworks/models/deepseek-v4-pro","fireworks_ai/deepseek-v4-pro","tencent/deepseek-v4-pro"],"hf_likes":2540,"hf_downloads":78864,"hf_downloads_all_time":78864,"hf_trending_score":2449,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"deepseek-v4-pro","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.435,"max_input_per_1m":2.4,"min_output_per_1m":0.87,"max_output_per_1m":4.8,"min_cache_read_per_1m":0.003625,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["deepseek","other/tencent"],"provider_count":8},"providers":[],"regions":[],"region_info":{}}},{"id":"zhipu-glm-5-2","name":"glm-5-2","display_name":"GLM-5.2","description":"Z.AI's flagship long-horizon autonomous LLM capable of sustained multi-hour task execution with agentic reasoning.","creator":"zhipu","family":"glm","tier":"","version":"5-2","type":"language","size_in_bn":753.33,"modalities":{"input":["text"],"output":["text"]},"context_window":1048576,"max_output_tokens":131072,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-06-16","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":7,"ids":["accounts/fireworks/models/glm-5p2","accounts/fireworks/models/glm-5p2-fp8","cloudflare/@cf/zai-org/glm-5.2","dashscope/glm-5.2","fireworks_ai/accounts/fireworks/models/glm-5p2","fireworks_ai/glm-5p2","glm-5-2","glm-5-2-non-reasoning","glm-5.2","huggingface-llm-glm-5-2-fp8","z-ai/glm-5.2","z-ai/glm-5.2:batch","zai-glm-5-2","zai-org/glm-5.2","zai-org/GLM-5.2","zai/glm-5.2","zhipu-glm-5-2"],"hf_likes":692,"hf_downloads":0,"hf_downloads_all_time":0,"hf_trending_score":675,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"zhipu-glm-5-2","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.63,"max_input_per_1m":2.052,"min_output_per_1m":1.98,"max_output_per_1m":6.27,"min_cache_read_per_1m":0.0945,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":6},"providers":[],"regions":[],"region_info":{}}},{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","display_name":"DeepSeek V4 Flash","description":"An efficiency-optimized Mixture-of-Experts LLM from DeepSeek with 284B total and 13B activated parameters, supporting a 1M-token context window with reasoning and tool-use capabilities.","creator":"deepseek","family":"v4","tier":"flash","version":null,"type":"language","size_in_bn":158.069,"modalities":{"input":["image","pdf","text"],"output":["text"]},"context_window":1048576,"max_output_tokens":393216,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"DeepSeek","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":true,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-04-24","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":12,"ids":["~deepseek/deepseek-v4-flash-latest","accounts/fireworks/models/deepseek-v4-flash","azure_ai/deepseek-v4-flash","dashscope/deepseek-v4-flash","deepseek-ai/DeepSeek-V4-Flash","deepseek-v4-flash","deepseek-v4-flash-high","deepseek-v4-flash-non-reasoning","deepseek-v4-flash(1)","deepseek-v4-flash*","deepseek/deepseek-v4-flash","deepseek/deepseek-v4-flash:free","fireworks_ai/accounts/fireworks/models/deepseek-v4-flash","fireworks_ai/deepseek-v4-flash","libertai/deepseek-v4-flash","pinstripes/ps/deepseek-v4-flash","tencent/deepseek-v4-flash","tensormesh/deepseek-ai/DeepSeek-V4-Flash"],"hf_likes":649,"hf_downloads":25391,"hf_downloads_all_time":25391,"hf_trending_score":639,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"deepseek-v4-flash","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.079996,"max_input_per_1m":0.25,"min_output_per_1m":0.2,"max_output_per_1m":1.75,"min_cache_read_per_1m":0.0028,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":11},"providers":[],"regions":[],"region_info":{}}},{"id":"minimax-m3","name":"minimax-m3","display_name":"MiniMax M3","description":"A multimodal foundation model supporting text, image, and video inputs with a 1M-token context window, designed for long-horizon agentic tasks, coding, and reasoning.","creator":"minimax","family":"m3","tier":"","version":null,"type":"language","size_in_bn":427.04,"modalities":{"input":["image","pdf","text","video"],"output":["text"]},"context_window":1048576,"max_output_tokens":512000,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-05-31","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":6,"ids":["accounts/fireworks/models/minimax-m3","fireworks_ai/accounts/fireworks/models/minimax-m3","fireworks_ai/minimax-m3","minimax-m3","minimax/minimax-m3","minimax/MiniMax-M3","minimax/minimax-m3:batch","MiniMaxAI/MiniMax-M3"],"hf_likes":320,"hf_downloads":442,"hf_downloads_all_time":442,"hf_trending_score":314,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"minimax-m3","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.15,"max_input_per_1m":0.3,"min_output_per_1m":0.6,"max_output_per_1m":1.2,"min_cache_read_per_1m":0.03,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":5},"providers":[],"regions":[],"region_info":{}}},{"id":"moonshot-kimi-k2-6","name":"kimi-k2-6","display_name":"Kimi K2.6","description":"An open-source native multimodal agentic LLM specializing in long-horizon coding, coding-driven design, autonomous execution, and swarm-based task orchestration.","creator":"moonshot","family":"kimi_k25","tier":"","version":"k2-6","type":"language","size_in_bn":1058.589,"modalities":{"input":["image","pdf","text","video"],"output":["text"]},"context_window":262144,"max_output_tokens":32768,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-04-20","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":9,"ids":["accounts/fireworks/models/kimi-k2p6","azure_ai/kimi-k2.6","cloudflare/@cf/moonshotai/kimi-k2.6","fireworks_ai/accounts/fireworks/models/kimi-k2p6","fireworks_ai/kimi-k2p6","kimi-k2-6","kimi-k2-6-non-reasoning","kimi-k2.6","moonshot-kimi-k2-6","moonshot/kimi-k2.6","moonshotai/kimi-k2.6","moonshotai/Kimi-K2.6","moonshotai/kimi-k2.6:free","tensormesh/moonshotai/Kimi-K2.6"],"hf_likes":568,"hf_downloads":8241,"hf_downloads_all_time":8241,"hf_trending_score":560,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"moonshot-kimi-k2-6","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.8,"max_input_per_1m":1.2,"min_output_per_1m":3.4,"max_output_per_1m":4.5,"min_cache_read_per_1m":0.16,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["huggingface"],"provider_count":8},"providers":[],"regions":[],"region_info":{}}},{"id":"moonshot-kimi-k2-code","name":"kimi-k2-code","display_name":"Kimi K2.7 Code","description":"A coding-focused agentic LLM built on Kimi K2.6, optimized for long-horizon real-world coding tasks and end-to-end task completion with tool-use and vision support.","creator":"moonshot","family":"kimi_k25","tier":"","version":"k2","type":"language","size_in_bn":1058.589,"modalities":{"input":["image","pdf","text","video"],"output":["text"]},"context_window":262144,"max_output_tokens":32768,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-06-12","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":6,"ids":["accounts/fireworks/models/kimi-k2p7-code","cloudflare/@cf/moonshotai/kimi-k2.7-code","dashscope/kimi-k2.7-code","fireworks_ai/accounts/fireworks/models/kimi-k2p7-code","fireworks_ai/kimi-k2p7-code","kimi-k2-7-code","kimi-k2.7-code","moonshot-kimi-k2-code","moonshotai/kimi-k2.7-code","moonshotai/Kimi-K2.7-Code"],"hf_likes":390,"hf_downloads":0,"hf_downloads_all_time":0,"hf_trending_score":383,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"moonshot-kimi-k2-code","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.67,"max_input_per_1m":0.95,"min_output_per_1m":3.4,"max_output_per_1m":4,"min_cache_read_per_1m":0.15,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":5},"providers":[],"regions":[],"region_info":{}}},{"id":"zhipu-glm-5-1","name":"glm-5-1","display_name":"GLM-5.1","description":"Z AI's next-generation flagship agentic LLM with significantly stronger coding capabilities, vision support, and file input, achieving top performance on SWE-Bench.","creator":"zhipu","family":"glm_moe_dsa","tier":"","version":"5-1","type":"language","size_in_bn":753.864,"modalities":{"input":["image","pdf","text"],"output":["text"]},"context_window":204800,"max_output_tokens":131072,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-04-07","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":6,"ids":["accounts/fireworks/models/glm-5p1","dashscope/glm-5.1","fireworks_ai/accounts/fireworks/models/glm-5p1","fireworks_ai/glm-5p1","glm-5-1","glm-5-1-non-reasoning","glm-5.1","huggingface-llm-glm-5-1-fp8","openrouter/z-ai/glm-5.1","z-ai/glm-5.1","zai-org/glm-5.1","zai-org/GLM-5.1","zai/glm-5.1","zhipu-glm-5-1"],"hf_likes":1449,"hf_downloads":147738,"hf_downloads_all_time":147738,"hf_trending_score":214,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"zhipu-glm-5-1","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":1.4,"max_input_per_1m":1.4,"min_output_per_1m":4.4,"max_output_per_1m":4.4,"min_cache_read_per_1m":0.26,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["alibaba_qwen","fireworks_ai","openrouter","vercel_ai_gateway","z_ai"],"provider_count":5},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-nemotron-3-ultra-550b-a55b","name":"nemotron-3-ultra-550b-a55b","display_name":"Nemotron 3 Ultra 550B A55B","description":"A 550B-parameter mixture-of-experts Nemotron model with 55B active parameters, built for frontier-scale reasoning, tool use, and agentic tasks.","creator":"nvidia","family":"nemotron","tier":"ultra","version":"3","type":"language","size_in_bn":550,"modalities":{"input":["text"],"output":["text"]},"context_window":1000000,"max_output_tokens":65536,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-06-04","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":4,"ids":["huggingface-reasoning-nvidia-nemotron-3-ultra-550b-a55b-nvfp4","nvidia-nemotron-3-ultra-550b-a55b","nvidia/nemotron-3-ultra-550b-a55b","nvidia/nemotron-3-ultra-550b-a55b:batch","nvidia/nemotron-3-ultra-550b-a55b:free","nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"],"hf_likes":225,"hf_downloads":111067,"hf_downloads_all_time":111067,"hf_trending_score":22,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nvidia-nemotron-3-ultra-550b-a55b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.3,"max_input_per_1m":0.6,"min_output_per_1m":1.8,"max_output_per_1m":3.6,"min_cache_read_per_1m":0.1,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":3},"providers":[],"regions":[],"region_info":{}}},{"id":"minimax-m2-5","name":"minimax-m2-5","display_name":"MiniMax M2.5","description":"MiniMax's M2.5 generation MoE language model offering strong reasoning and tool-use performance for complex productivity and agent tasks.","creator":"minimax","family":"minimax_m2","tier":"","version":"5","type":"language","size_in_bn":228.704,"modalities":{"input":["text"],"output":["text"]},"context_window":1000000,"max_output_tokens":196608,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-02-12","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":9,"ids":["accounts/fireworks/models/minimax-m2p5","baseten/MiniMaxAI/MiniMax-M2.5","bedrock/ap-northeast-1/minimax.minimax-m2.5","bedrock/ap-south-1/minimax.minimax-m2.5","bedrock/ap-southeast-2/minimax.minimax-m2.5","bedrock/ap-southeast-3/minimax.minimax-m2.5","bedrock/eu-central-1/minimax.minimax-m2.5","bedrock/eu-north-1/minimax.minimax-m2.5","bedrock/eu-south-1/minimax.minimax-m2.5","bedrock/eu-west-1/minimax.minimax-m2.5","bedrock/eu-west-2/minimax.minimax-m2.5","bedrock/sa-east-1/minimax.minimax-m2.5","bedrock/us-east-1/minimax.minimax-m2.5","bedrock/us-east-2/minimax.minimax-m2.5","bedrock/us-west-2/minimax.minimax-m2.5","huggingface-llm-minimax-m2-5","minimax-m2-5","minimax.minimax-m2.5","minimax/minimax-m2.5","minimax/MiniMax-M2.5","minimax/minimax-m2.5:free","MiniMaxAI/MiniMax-M2.5","openrouter/minimax/minimax-m2.5","tensormesh/MiniMaxAI/MiniMax-M2.5","wandb/MiniMaxAI/MiniMax-M2.5"],"hf_likes":1461,"hf_downloads":928266,"hf_downloads_all_time":1586202,"hf_trending_score":13,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"minimax-m2-5","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.22,"max_input_per_1m":0.3,"min_output_per_1m":0.9,"max_output_per_1m":1.2,"min_cache_read_per_1m":0.03,"min_cache_write_per_1m":0.375,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":8},"providers":[],"regions":[],"region_info":{}}},{"id":"alibaba-qwen3-5-397b-a17b","name":"qwen3-5-397b-a17b","display_name":"Qwen3.5 397B A17B","description":"Alibaba's largest Qwen3.5 MoE model with 397B total parameters and 17B activated per token, targeting maximum capability for complex reasoning and generation.","creator":"alibaba","family":"qwen3_5_moe","tier":"","version":null,"type":"language","size_in_bn":397,"modalities":{"input":["image","text","video"],"output":["text"]},"context_window":262144,"max_output_tokens":65536,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Qwen3","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-02-16","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":6,"ids":["accounts/fireworks/models/qwen3p5-397b-a17b","alibaba-qwen3-5-397b-a17b","openrouter/qwen/qwen3.5-397b-a17b","qwen/qwen3.5-397b-a17b","Qwen/Qwen3.5-397B-A17B","qwen3-5-397b-a17b","qwen3-5-397b-a17b-non-reasoning","qwen3.5-397b-a17b","scaleway/qwen/qwen3.5-397b-a17b","together_ai/Qwen/Qwen3.5-397B-A17B"],"hf_likes":1462,"hf_downloads":710153,"hf_downloads_all_time":2631436,"hf_trending_score":11,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen3-5-397b-a17b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.5,"max_input_per_1m":0.71,"min_output_per_1m":3.6,"max_output_per_1m":4.25,"min_cache_read_per_1m":0.3,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":5},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-nemotron-super-3-120b-a12b","name":"nvidia-nemotron-super-3-120b-a12b","display_name":"Nemotron Super 3 120B A12B","description":"A 120B-parameter hybrid MoE Nemotron Super 3 model with 12B active parameters, optimized by NVIDIA for compute-efficient reasoning in specialized agentic systems.","creator":"nvidia","family":"nemotron_h","tier":"","version":"3","type":"language","size_in_bn":120,"modalities":{"input":["text"],"output":["text"]},"context_window":1000000,"max_output_tokens":32000,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-03-11","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":3,"ids":["accounts/fireworks/models/nvidia-nemotron-3-super-120b-a12b-fp8","accounts/fireworks/models/nvidia-nemotron-3-super-120b-a12b-nvfp4","huggingface-llm-nvidia-nemotron-3-super-120b-a12b-bf16","nvidia-nemotron-3-super-120b-a12b","nvidia-nemotron-super-3-120b-a12b","nvidia/nemotron-3-super-120b-a12b","nvidia/nemotron-3-super-120b-a12b:free","nvidia/NVIDIA-Nemotron-3-Super-120B-A12B"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nvidia-nemotron-super-3-120b-a12b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.085,"max_input_per_1m":0.15,"min_output_per_1m":0.4,"max_output_per_1m":0.65,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":2},"providers":[],"regions":[],"region_info":{}}},{"id":"openai-gpt-oss-120b","name":"gpt-oss-120b","display_name":"GPT OSS 120B","description":"A 120-billion-parameter open-weights GPT model from OpenAI designed for reasoning-intensive tasks with implicit caching support.","creator":"openai","family":"gpt_oss","tier":"","version":null,"type":"language","size_in_bn":120,"modalities":{"input":["image","text"],"output":["text"]},"context_window":131072,"max_output_tokens":131072,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2024-06","training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"GPT","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":true,"reasoning":true,"web_search":true,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-08-05","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":24,"ids":["@cf/openai/gpt-oss-120b","accounts/fireworks/models/gpt-oss-120b","azure_ai/gpt-oss-120b","baseten/openai/gpt-oss-120b","bedrock_mantle/openai.gpt-oss-120b","cerebras/gpt-oss-120b","cloudflare/@cf/openai/gpt-oss-120b","crusoe/openai/gpt-oss-120b","databricks/databricks-gpt-oss-120b","deepinfra/openai/gpt-oss-120b","fireworks_ai/accounts/fireworks/models/gpt-oss-120b","fireworks_ai/gpt-oss-120b","gpt-oss-120b","gpt-oss-120b-low","gpt-oss-120b-maas","groq/openai/gpt-oss-120b","lemonade/gpt-oss-120b-mxfp-GGUF","novita/openai/gpt-oss-120b","ollama/gpt-oss:120b-cloud","openai-gpt-oss-120b","openai-reasoning-gpt-oss-120b","openai.gpt-oss-120b-1:0","openai/gpt-oss-120b","openai/gpt-oss-120b:free","openrouter/openai/gpt-oss-120b","ovhcloud/gpt-oss-120b","publishers/google/models/gpt-oss-120b-maas","replicate/openai/gpt-oss-120b","sambanova/gpt-oss-120b","scaleway/openai/gpt-oss-120b","tensormesh/openai/gpt-oss-120b","together_ai/openai/gpt-oss-120b","vertex_ai/openai/gpt-oss-120b-maas","wandb/openai/gpt-oss-120b","watsonx/openai/gpt-oss-120b"],"hf_likes":4719,"hf_downloads":3524674,"hf_downloads_all_time":32348365,"hf_trending_score":25,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"openai-gpt-oss-120b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.03,"max_input_per_1m":15,"min_output_per_1m":0.17,"max_output_per_1m":60,"min_cache_read_per_1m":0.015,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":23},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-nemotron-lightning-3-5","name":"nemotron-lightning-3-5","display_name":"Nemotron Lightning 3.5","description":"NVIDIA's fast open model targeting always-on agents and high-volume specialized tasks with strong agentic capability.","creator":"nvidia","family":"nemotron","tier":"","version":"3-5","type":"language","size_in_bn":31.578,"modalities":{"input":["text"],"output":["text"]},"context_window":1000000,"max_output_tokens":262144,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2026-08-11","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":3,"ids":["deepinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning","nemotron-3-5-lightning","nvidia-nemotron-lightning-3-5","nvidia/nemotron-3.5-lightning","nvidia/nemotron-3.5-lightning:free","nvidia/NVIDIA-Nemotron-3.5-Lightning","openrouter/nvidia/nemotron-3.5-lightning"],"hf_likes":120,"hf_downloads":15740,"hf_downloads_all_time":15740,"hf_trending_score":120,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nvidia-nemotron-lightning-3-5","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.05,"max_input_per_1m":0.1,"min_output_per_1m":0.2,"max_output_per_1m":0.25,"min_cache_read_per_1m":0.05,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["deepinfra"],"provider_count":2},"providers":[],"regions":[],"region_info":{}}},{"id":"alibaba-qwen3-235b-a22b-instruct","name":"qwen3-235b-a22b-instruct","display_name":"Qwen3 235B A22B Instruct","description":"An instruction-tuned update of the Qwen3 235B A22B MoE model with significant improvements in instruction following, logical reasoning, and general capabilities.","creator":"alibaba","family":"qwen3_moe","tier":"","version":null,"type":"language","size_in_bn":235,"modalities":{"input":["text"],"output":["text"]},"context_window":262144,"max_output_tokens":16384,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":null,"earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":12,"ids":["accounts/fireworks/models/qwen3-235b-a22b-instruct-2507","alibaba-qwen3-235b-a22b-instruct","crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507","deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507","fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-instruct-2507","novita/qwen/qwen3-235b-a22b-instruct-2507","qwen/qwen3-235b-a22b-instruct-2507","Qwen/Qwen3-235B-A22B-Instruct-2507","qwen3-235b-a22b-instruct","qwen3-235b-a22b-instruct-2507","replicate/qwen/qwen3-235b-a22b-instruct-2507","scaleway/qwen/qwen3-235b-a22b-instruct-2507","wandb/Qwen/Qwen3-235B-A22B-Instruct-2507"],"hf_likes":773,"hf_downloads":150781,"hf_downloads_all_time":1182969,"hf_trending_score":1,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen3-235b-a22b-instruct","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.09,"max_input_per_1m":10,"min_output_per_1m":0.58,"max_output_per_1m":10,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["deepinfra","huggingface","novita"],"provider_count":11},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-nemotron-nano-3-30b-a3b","name":"nvidia-nemotron-nano-3-30b-a3b","display_name":"Nemotron Nano 3 30B A3B","description":"A 30B-parameter hybrid MoE Nemotron Nano 3 model with 3B active parameters, combining Mamba-Transformer architecture for efficient reasoning and agentic tasks.","creator":"nvidia","family":"nemotron_h","tier":"","version":"3","type":"language","size_in_bn":30,"modalities":{"input":["text"],"output":["text"]},"context_window":262144,"max_output_tokens":228000,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-12-14","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":3,"ids":["accounts/fireworks/models/nemotron-nano-3-30b-a3b","nemotron-3-nano-30b-a3b","nvidia-nemotron-3-nano-30b-a3b","nvidia-nemotron-3-nano-30b-a3b-reasoning","nvidia-nemotron-nano-3-30b-a3b","nvidia/nemotron-3-nano-30b-a3b","nvidia/Nemotron-3-Nano-30B-A3B","nvidia/nemotron-3-nano-30b-a3b:free","nvidia/nvidia-nemotron-3-nano-30b-a3b-bf16"],"hf_likes":767,"hf_downloads":1129029,"hf_downloads_all_time":6131226,"hf_trending_score":8,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nvidia-nemotron-nano-3-30b-a3b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.05,"max_input_per_1m":0.05,"min_output_per_1m":0.2,"max_output_per_1m":0.24,"min_cache_read_per_1m":0.025,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter","vercel_ai_gateway"],"provider_count":2},"providers":[],"regions":[],"region_info":{}}},{"id":"alibaba-qwen3-30b-a3b-instruct","name":"qwen3-30b-a3b-instruct","display_name":"Qwen3 30B A3B Instruct","description":"An instruction-tuned Qwen3 MoE model with 30B total and 3B active parameters, optimized for text generation and instruction-following tasks.","creator":"alibaba","family":"qwen3_moe","tier":"","version":null,"type":"language","size_in_bn":30,"modalities":{"input":["text"],"output":["text"]},"context_window":262144,"max_output_tokens":32000,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2025-06-30","training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":"Qwen3","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-07-29","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":4,"ids":["accounts/fireworks/models/qwen3-30b-a3b-instruct-2507","accounts/fireworks/models/qwen3-30b-a3b-instruct-2507-eagle3-v0","alibaba-qwen3-30b-a3b-instruct","fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-instruct-2507","huggingface-reasoning-qwen3-30b-a3b-instruct-2507","qwen/qwen3-30b-a3b-instruct-2507","qwen3-30b-a3b-instruct","qwen3-30b-a3b-instruct-2507"],"hf_likes":801,"hf_downloads":983022,"hf_downloads_all_time":10380022,"hf_trending_score":2,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen3-30b-a3b-instruct","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.04815,"max_input_per_1m":0.5,"min_output_per_1m":0.19305,"max_output_per_1m":0.8,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":3},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-cosmos-3-super-reasoner","name":"cosmos-3-super-reasoner","display_name":"Cosmos 3 Super Reasoner","description":"A large-scale physical AI reasoning model in NVIDIA's Cosmos series, designed for world-model simulation and advanced multi-step reasoning tasks.","creator":"nvidia","family":"cosmos","tier":"super","version":"3","type":"language","size_in_bn":null,"modalities":{"input":["text"],"output":["text"]},"context_window":null,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":false,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":null,"earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":1,"ids":["nvidia-cosmos-3-super-reasoner"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18"},{"id":"google-gemma-3-27b-instruct","name":"gemma-3-27b-instruct","display_name":"Gemma 3 27B Instruct","description":"An instruction-tuned 27B Gemma 3 LLM with multimodal vision-language input and 128k context window.","creator":"google","family":"gemma3","tier":"","version":"3","type":"language","size_in_bn":27,"modalities":{"input":["image","text"],"output":["text"]},"context_window":262144,"max_output_tokens":131072,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2024-08-31","training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":"Gemini","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-03-12","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":7,"ids":["accounts/fireworks/models/gemma-3-27b-it","deepinfra/google/gemma-3-27b-it","fireworks_ai/accounts/fireworks/models/gemma-3-27b-it","gemini/gemma-3-27b-it","google-gemma-3-27b-instruct","google.gemma-3-27b-it","google/gemma-3-27b-it","google/gemma-3-27b-it:free","huggingface-vlm-gemma-3-27b-instruct","nebius/google/gemma-3-27b-it","novita/google/gemma-3-27b-it","scaleway/google/gemma-3-27b-it"],"hf_likes":1956,"hf_downloads":567671,"hf_downloads_all_time":12733530,"hf_trending_score":2,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"google-gemma-3-27b-instruct","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.08,"max_input_per_1m":0.9,"min_output_per_1m":0.16,"max_output_per_1m":0.9,"min_cache_read_per_1m":0.04,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":6},"providers":[],"regions":[],"region_info":{}}},{"id":"nousresearch-hermes-4-405b","name":"hermes-4-405b","display_name":"Hermes 4 405B","description":"A 405B-parameter hybrid reasoning LLM from Nous Research built on Llama 3.1, capable of deliberate internal reasoning or direct response generation.","creator":"nousresearch","family":"hermes","tier":"","version":null,"type":"language","size_in_bn":405,"modalities":{"input":["text"],"output":["text"]},"context_window":131072,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2024-08-31","training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Other","capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":"2025-08-26","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":2,"ids":["nousresearch-hermes-4-405b","nousresearch/hermes-4-405b"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nousresearch-hermes-4-405b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":1,"max_input_per_1m":1,"min_output_per_1m":3,"max_output_per_1m":3,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":1},"providers":[],"regions":[],"region_info":{}}},{"id":"nousresearch-hermes-4-70b","name":"hermes-4-70b","display_name":"Hermes 4 70B","description":"A 70B-parameter hybrid reasoning LLM from Nous Research built on Llama 3.1, supporting both deliberate thinking and direct instruction-following modes.","creator":"nousresearch","family":"hermes","tier":"","version":null,"type":"language","size_in_bn":70,"modalities":{"input":["text"],"output":["text"]},"context_window":131072,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2024-08-31","training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Llama3","capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":"2025-08-26","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":2,"ids":["nousresearch-hermes-4-70b","nousresearch/hermes-4-70b"],"hf_likes":185,"hf_downloads":1335,"hf_downloads_all_time":41341,"hf_trending_score":0,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"nousresearch-hermes-4-70b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.13,"max_input_per_1m":0.13,"min_output_per_1m":0.4,"max_output_per_1m":0.4,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":1},"providers":[],"regions":[],"region_info":{}}},{"id":"nvidia-llama-3-1-nemotron-1-ultra-253b","name":"llama-3-1-nemotron-1-ultra-253b","display_name":"Llama 3.1 Nemotron 1 Ultra 253B","description":"A 253B-parameter ultra-scale LLM fine-tuned by NVIDIA on Llama 3.1, optimized for advanced reasoning and high-accuracy agentic tasks.","creator":"nvidia","family":"llama","tier":"ultra","version":"1","type":"language","size_in_bn":253,"modalities":{"input":["text"],"output":["text"]},"context_window":128000,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":false,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":null,"earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":1,"ids":["llama-3-1-nemotron-ultra-253b-v1-reasoning","nebius/nvidia/Llama-3.1-Nemotron-Ultra-253B-v1","nvidia-llama-3-1-nemotron-1-ultra-253b"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18"},{"id":"openbmb-minicpm-4-5-vl","name":"minicpm-4-5-vl","display_name":"MiniCPM-V 4.5 VL","description":"A vision-language variant of the MiniCPM 4.5 series, offering multimodal understanding in a compact, edge-deployable model.","creator":"openbmb","family":"minicpm","tier":"","version":"4-5","type":"language","size_in_bn":null,"modalities":{"input":["text"],"output":["text"]},"context_window":null,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":false,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":null,"earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":1,"ids":["openbmb-minicpm-4-5-vl"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18"},{"id":"nvidia-nemotron-nano-3-omni","name":"nemotron-nano-3-omni","display_name":"Nemotron Nano 3 Omni","description":"A compact omni-modal Nemotron model from NVIDIA's Nano tier, combining text, vision, and audio understanding in a small, efficient footprint.","creator":"nvidia","family":"nemotron","tier":"","version":"3","type":"language","size_in_bn":null,"modalities":{"input":["text"],"output":["text"]},"context_window":null,"max_output_tokens":null,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":false,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":null,"earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":1,"ids":["nvidia-nemotron-nano-3-omni"],"hf_likes":null,"hf_downloads":null,"hf_downloads_all_time":null,"hf_trending_score":null,"updated_at":"2026-08-14 08:03:18"},{"id":"alibaba-qwen2-5-vl-72b-instruct","name":"qwen2-5-vl-72b-instruct","display_name":"Qwen2.5 VL 72B Instruct","description":"A 72-billion-parameter multimodal vision-language LLM from Alibaba's Qwen2.5-VL series, delivering high-capacity image understanding and visual reasoning.","creator":"alibaba","family":"qwen2_5_vl","tier":"","version":null,"type":"language","size_in_bn":72,"modalities":{"input":["image","text"],"output":["text"]},"context_window":131072,"max_output_tokens":128000,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2024-06-30","training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":"Qwen","capabilities":{"function_calling":true,"parallel_function_calling":false,"structured_outputs":true,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-02-01","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":6,"ids":["accounts/fireworks/models/qwen2p5-vl-72b-instruct","alibaba-qwen2-5-vl-72b-instruct","fireworks_ai/accounts/fireworks/models/qwen2p5-vl-72b-instruct","nebius/Qwen/Qwen2.5-VL-72B-Instruct","novita/qwen/qwen2.5-vl-72b-instruct","ovhcloud/Qwen2.5-VL-72B-Instruct","qwen/qwen2.5-vl-72b-instruct","qwen2.5-vl-72b-instruct"],"hf_likes":609,"hf_downloads":103451,"hf_downloads_all_time":5812114,"hf_trending_score":1,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen2-5-vl-72b-instruct","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.25,"max_input_per_1m":1.01,"min_output_per_1m":0.75,"max_output_per_1m":1.01,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["openrouter"],"provider_count":5},"providers":[],"regions":[],"region_info":{}}},{"id":"alibaba-qwen3-embedding-8b","name":"qwen3-embedding-8b","display_name":"Qwen3 Embedding 8B","description":"An 8B-parameter text embedding model from the Qwen3 series with strong multilingual capabilities and long-context support for retrieval and ranking tasks.","creator":"alibaba","family":"qwen3","tier":"","version":null,"type":"embedding","size_in_bn":8,"modalities":{"input":["text"],"output":["embedding"]},"context_window":40960,"max_output_tokens":4096,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":null,"training_data_cutoff":null,"supported_reasoning_efforts":[],"tokenizer":null,"capabilities":{"function_calling":false,"parallel_function_calling":false,"structured_outputs":false,"prompt_caching":false,"reasoning":false,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":false,"adaptive_reasoning":false},"release_date":"2025-06-05","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":5,"ids":["accounts/fireworks/models/qwen3-embedding-8b","alibaba-qwen3-embedding-8b","alibaba/qwen3-embedding-8b","llamagate/qwen3-embedding-8b","novita/qwen/qwen3-embedding-8b","Qwen/Qwen3-Embedding-8B","scaleway/qwen/qwen3-embedding-8b"],"hf_likes":656,"hf_downloads":1900241,"hf_downloads_all_time":10616532,"hf_trending_score":9,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen3-embedding-8b","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.02,"max_input_per_1m":0.1,"min_output_per_1m":null,"max_output_per_1m":null,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["other/llamagate"],"provider_count":4},"providers":[],"regions":[],"region_info":{}}},{"id":"alibaba-qwen3-next-80b-a3b-thinking","name":"qwen3-next-80b-a3b-thinking","display_name":"Qwen3 Next 80B A3B Thinking","description":"A thinking-mode variant of the Qwen3 Next MoE model with 80B total and 3B activated parameters, featuring hybrid attention for extended chain-of-thought reasoning.","creator":"alibaba","family":"qwen3_next","tier":"","version":null,"type":"language","size_in_bn":80,"modalities":{"input":["text"],"output":["text"]},"context_window":262144,"max_output_tokens":65536,"tool_use_system_prompt_tokens":0,"output_vector_sizes":[],"knowledge_cutoff":"2025-09-30","training_data_cutoff":null,"supported_reasoning_efforts":["default"],"tokenizer":"Qwen3","capabilities":{"function_calling":true,"parallel_function_calling":true,"structured_outputs":true,"prompt_caching":false,"reasoning":true,"web_search":false,"computer_use":false,"code_execution":false,"file_search":false,"url_context":false,"assistant_prefill":false,"native_structured_output":true,"adaptive_reasoning":false},"release_date":"2025-09-11","earliest_deprecation_date":null,"deprecated":false,"has_pricing":true,"provider_count":10,"ids":["accounts/fireworks/models/qwen3-next-80b-a3b-thinking","alibaba-qwen3-next-80b-a3b-thinking","alibaba/qwen3-next-80b-a3b-thinking","dashscope/qwen3-next-80b-a3b-thinking","deepinfra/Qwen/Qwen3-Next-80B-A3B-Thinking","fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-thinking","novita/qwen/qwen3-next-80b-a3b-thinking","qwen/qwen3-next-80b-a3b-thinking","qwen3-next-80b-a3b-thinking","together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking"],"hf_likes":487,"hf_downloads":32460,"hf_downloads_all_time":2304469,"hf_trending_score":0,"updated_at":"2026-08-14 08:03:18","pricing":{"model_id":"alibaba-qwen3-next-80b-a3b-thinking","currency":"USD","exchange_rate":1,"exchange_rate_date":"2026-08-14","ingestion_date":"2026-08-14","summary":{"currency":"USD","min_input_per_1m":0.14,"max_input_per_1m":0.9,"min_output_per_1m":0.9,"max_output_per_1m":1.5,"min_cache_read_per_1m":null,"min_cache_write_per_1m":null,"min_reasoning_per_1m":null,"cheapest_providers":["deepinfra"],"provider_count":9},"providers":[],"regions":[],"region_info":{}}}],"pagination":{"page_size":50,"has_next":false,"next_token":null,"total_count":27},"meta":{"updated_at":"2026-08-14","request_id":"5c0fb16a-8ce1-4a60-a02c-633cd39176eb","execution_ms":9}}