{"data":[{"schema_version":"2.4","id":"bge-reranker-v2-m3","name":"BGE M3","created":1789490421,"description":"BAAI bge-reranker-v2-m3 is a 0.6B-parameter multilingual reranking model built on bge-m3. It scores query-document relevance for retrieval and RAG pipelines with an 8,192-token input window and fast inference.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"BAAI\/bge-reranker-v2-m3","quantization":"fp16","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":8192,"unit":"token"}}}],"output_modalities":[{"type":"rerank","supported_parameters":{"top_n":{"type":"integer","min":1,"max":100},"return_documents":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000005"}]}]},{"schema_version":"2.4","id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","created":1786979709,"description":"DeepSeek's official V4 Flash release with substantially enhanced agentic capabilities and a built-in speculative decoding module. A 304B-parameter MoE with ~13B active per token, it outperforms DeepSeek-V4-Pro (Preview) on agentic benchmarks at a fraction of the compute, with a 1M-token context window.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"deepseek-ai\/DeepSeek-V4-Flash-0731","quantization":"fp8","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.0000001"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000025"}]}]},{"schema_version":"2.4","id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","created":1789449526,"description":"DeepSeek V4.1 Flash is a mixture-of-experts model with 552B backbone parameters, a one-million-token context window, and support for reasoning and tool calling.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"deepseek-ai\/DeepSeek-V4.1-Flash","quantization":"mxfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000015"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.0000006"}]}]},{"schema_version":"2.4","id":"glm-5.2","name":"GLM 5.2","created":1788472794,"description":"GLM-5.2 is Z.ai's flagship model for long-horizon tasks, sustaining agentic work across a solid 1M-token context. It offers stronger coding with flexible thinking-effort levels and an improved MTP layer for faster speculative decoding, under an MIT license.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"nvidia\/GLM-5.2-NVFP4","quantization":"nvfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000075"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.000003"}]}]},{"schema_version":"2.4","id":"glm-5.3","name":"GLM 5.3","created":1788472794,"description":"GLM-5.3 shares its base model with GLM-5.2 and takes its gains from post-training, with substantially stronger complex coding and long-horizon agentic work. Served at NVFP4 across a 1M-token context.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"RadixArk\/GLM-5.3-NVFP4","quantization":"nvfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000075"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.000003"}]}]},{"schema_version":"2.4","id":"glm-5.3-flash","name":"GLM 5.3 Flash","created":1788837087,"description":"Z.ai's first natively multimodal GLM-5 model: a 320B-parameter MoE with ~18B active per token, combining sparse and linear attention for cheap long-context serving across a 1M-token window. Outperforms GLM-5.2 on coding and agentic work at a fraction of the price, with three-level reasoning-effort control, under an MIT license.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"RedHatAI\/GLM-5.3-Flash-NVFP4","quantization":"nvfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.0000001"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}},{"type":"video","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["video\/mp4","video\/webm"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000035"}]}]},{"schema_version":"2.4","id":"glm-5.x-menthol","name":"GLM 5","created":1788837089,"description":"https:\/\/huggingface.co\/zai-org\/GLM-5","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"zai-org\/GLM-5","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":202752,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.0000004"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":202752,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":202752,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000175"}]}]},{"schema_version":"2.4","id":"laguna-s-2.1","name":"Laguna S 2.1","created":1788837090,"description":"Laguna S 2.1 is a 118B total parameter Mixture-of-Experts model with 8B activated parameters per token, designed for agentic coding and long-horizon work. It sits between Laguna XS 2.1 (33B-A3B) and Laguna M.1 (225B-A23B) in the Laguna series and shares the family recipe: a token-choice router with softplus gating over 256 routed experts plus one shared expert, grouped-query attention, and interleaved full\/sliding-window attention.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"poolside\/Laguna-S-2.1-FP8","quantization":"fp8","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":1048576,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000009"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":1048576,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":1048576,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000018"}]}]},{"schema_version":"2.4","id":"minimax-m3","name":"MiniMax M3","created":1787633092,"description":"MiniMax M3 is a natively multimodal MoE model (~428B parameters, ~23B active) with a 1M-token context. MiniMax Sparse Attention delivers up to 9x prefill and 15x decode speedups at long context, with frontier-level performance on long-horizon coding and agentic benchmarks.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"MiniMaxAI\/MiniMax-M3","quantization":"fp8","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":524288,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.0000002"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}},{"type":"video","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["video\/mp4","video\/webm"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":524288,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":524288,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.0000009"}]}]},{"schema_version":"2.4","id":"muse-glimmer-30b","name":"Muse Glimmer 30B","created":1789531165,"description":"Muse Glimmer 30B is a dense multimodal model for reasoning, coding, and tool use, with a 131072-token context. Supports text and image input, and video processed as individual frames, with text output.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"nvidia\/Muse-Glimmer-30B-NVFP4","quantization":"nvfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":131072,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000025"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}},{"type":"video","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["video\/mp4","video\/webm"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":131072,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":131072,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.000001"}]}]},{"schema_version":"2.4","id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","created":1780598852,"description":"NVIDIA's Nemotron 3.5 Content Safety model is a small language model fine-tuned from Gemma-3-4B-it for multimodal, multilingual content moderation. It classifies prompts, images, and responses against a standard safety taxonomy or a user-supplied custom policy, with optional reasoning traces.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"nvidia\/Nemotron-3.5-Content-Safety","quantization":"bf16","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":131072,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000005"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":131072,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":131072,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000015"}]}]},{"schema_version":"2.4","id":"qwen3.8-27b","name":"Qwen 3.8 27B","created":1787624247,"description":"Qwen3.8-27B is the most capable compact dense model in the Qwen open-model family: a native vision-language model that understands images and video, with flexible thinking control and substantial gains in coding, research, and long-horizon agentic tasks.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"Qwen\/Qwen3.8-27B-FP8","quantization":"fp8","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":262144,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.00000015"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}},{"type":"video","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["video\/mp4","video\/webm"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":262144,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":262144,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.000001"}]}]},{"schema_version":"2.4","id":"qwen3.8-flash-next","name":"Qwen 3.8 Flash Next","created":1789485493,"description":"Qwen3.8 Flash Next with NVFP4 weights, a 262,144-token context window, reasoning, and tool calling.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"nvidia\/Qwen3.8-Flash-Next-NVFP4","quantization":"nvfp4","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":262144,"unit":"token"}},"pricing":[{"type":"prompt","unit":"token","cost_usd":"0.0000001"}]},{"type":"image","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["image\/png","image\/jpeg","image\/webp","image\/gif"]}}},{"type":"video","supported_inputs":{"sources":{"type":"enum","values":["url","base64"]},"formats":{"type":"enum","values":["video\/mp4","video\/webm"]}}}],"output_modalities":[{"type":"text","streaming":true,"max_length":{"value":262144,"unit":"token"},"supported_parameters":{"temperature":{"type":"range","min":0,"max":2},"top_p":{"type":"range","min":0,"max":1},"frequency_penalty":{"type":"range","min":-2,"max":2},"presence_penalty":{"type":"range","min":-2,"max":2},"max_tokens":{"type":"integer","min":1,"max":262144,"unit":"token"},"stop":{"type":"array","items":{"type":"unknown"},"max_items":4},"seed":{"type":"boolean"},"continue_final_message":{"type":"boolean"},"tools":{"type":"boolean"},"reasoning":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.0000002"}]}]},{"schema_version":"2.4","id":"vultron-retriever-core-qwen3.5-4.5b","name":"Vultron Retriever Core 4.5B","created":1783002744,"description":"Vultr's Vultron Retriever Core is a 4.5B-parameter reranking model built on Qwen3.5. It scores query-document relevance for retrieval and RAG pipelines, balancing accuracy and throughput with a 262k-token context.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"vultr\/VultronRetrieverCore-Qwen3.5-4.5B","quantization":"bf16","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":262144,"unit":"token"}}}],"output_modalities":[{"type":"rerank","supported_parameters":{"top_n":{"type":"integer","min":1,"max":100},"return_documents":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.0000001"}]}]},{"schema_version":"2.4","id":"vultron-retriever-flash-qwen3.5-0.8b","name":"Vultron Retriever Flash 0.8B","created":1783002800,"description":"Vultr's Vultron Retriever Flash is a lightweight 0.8B-parameter reranking model built on Qwen3.5, optimized for low-latency, high-volume query-document relevance scoring in retrieval and RAG pipelines.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"vultr\/VultronRetrieverFlash-Qwen3.5-0.8B","quantization":"bf16","input_modalities":[{"type":"text","supported_inputs":{"max_context_length":{"value":262144,"unit":"token"}}}],"output_modalities":[{"type":"rerank","supported_parameters":{"top_n":{"type":"integer","min":1,"max":100},"return_documents":{"type":"boolean"}},"pricing":[{"type":"completion","unit":"token","cost_usd":"0.00000005"}]}]},{"schema_version":"2.4","id":"z-image-turbo","name":"Z-Image Turbo","created":1787526201,"description":"Z-Image Turbo is Tongyi-MAI's efficient 6B-parameter single-stream diffusion transformer for text-to-image generation, producing high-quality images in as few as 8 steps for very low latency.","datacenters":[{"country_code":"US","region":"atl"}],"hugging_face_id":"Tongyi-MAI\/Z-Image-Turbo","quantization":"bf16","input_modalities":[{"type":"text"}],"output_modalities":[{"type":"image","streaming":false,"supported_parameters":{"n":{"type":"integer","min":1,"max":4},"size":{"type":"enum","values":["512x512","768x768","1024x1024"]},"response_format":{"type":"enum","values":["b64_json","url"]},"steps":{"type":"integer","min":1,"max":50,"default":8},"guidance_scale":{"type":"range","min":0,"max":20,"default":0}},"pricing":[{"type":"completion","unit":"megapixel","cost_usd":"0.02"}]}]}]}