{"data":[{"model_group":"apertus-70b","providers":["hosted_vllm"],"max_tokens":60000.0,"quantization":"FP16","max_input_tokens":30000.0,"max_output_tokens":30000.0,"input_cost_per_token":4e-07,"output_cost_per_token":2.1e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Supports tool use and agentic workflows. Apache 2.0 licensed, EU AI Act compliant.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"brick-complexity-pro","providers":["hosted_vllm"],"max_tokens":120000.0,"quantization":null,"max_input_tokens":100000.0,"max_output_tokens":15000.0,"input_cost_per_token":1e-07,"output_cost_per_token":4e-07,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Brick-complexity-pro is a model for complexity extraction from a query","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"brick-v1-beta","providers":["hosted_vllm"],"max_tokens":120000.0,"quantization":null,"max_input_tokens":100000.0,"max_output_tokens":15000.0,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"BETA - NOT FOR PRODUCTION - Bricks is a semantic router offered by Regolo.ai that analyzes each request and directs it to the most suitable model, optimizing costs and performance. Its cost equals the backend model's cost, with no additional overhead.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"deepseek-ocr-2","providers":["hosted_vllm"],"max_tokens":8000.0,"quantization":"FP16","max_input_tokens":4000.0,"max_output_tokens":4000.0,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"ocr","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":false,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"DeepSeek OCR 2 is a high‑accuracy optical character recognition model designed to extract text from complex visual inputs such as documents, screenshots, receipts, and natural scenes. It combines vision‑language modeling with efficient visual encoders to achieve superior recognition of multi‑language and multi‑layout text.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":0.02,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"faster-whisper-large-v3","providers":["hosted_vllm"],"max_tokens":null,"quantization":null,"max_input_tokens":null,"max_output_tokens":null,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"audio_transcription","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":false,"supports_audio_input":true,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"High-performance speech-to-text model based on OpenAI’s Whisper large-v3 architecture, optimized for fast and efficient inference using CTranslate2. It supports multilingual transcription and translation with high accuracy, especially for long-form audio and noisy environments","input_cost_per_second":0.00015,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"gemma4-31b","providers":["hosted_vllm"],"max_tokens":200000.0,"quantization":"FP16","max_input_tokens":100000.0,"max_output_tokens":100000.0,"input_cost_per_token":4e-07,"output_cost_per_token":2.1e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":true,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Advanced 31B multimodal model with native function-calling, reasoning capabilities, and extended context window of 256K tokens. Handles text, s, and video for diverse applications.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"glm5.2","providers":["hosted_vllm"],"max_tokens":200000.0,"quantization":"FP8","max_input_tokens":96000.0,"max_output_tokens":96000.0,"input_cost_per_token":2e-06,"output_cost_per_token":5.2e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"GLM-5.2 is a 753B-parameter sparse Mixture-of-Experts language model built for long-horizon tasks and advanced coding","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"gpt-oss-120b","providers":["hosted_vllm"],"max_tokens":120000.0,"quantization":"MXFP4","max_input_tokens":30000.0,"max_output_tokens":90000.0,"input_cost_per_token":1e-06,"output_cost_per_token":4.2e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":true,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"GPT-OSS-120B is an open-weight 117B-parameter Mixture-of-Experts model by OpenAI, using only 5.1B active parameters per token. Supports reasoning, chain-of-thought, tool use, and fine-tuning.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"gpt-oss-20b","providers":["hosted_vllm"],"max_tokens":120000.0,"quantization":"MXFP4","max_input_tokens":30000.0,"max_output_tokens":90000.0,"input_cost_per_token":1e-07,"output_cost_per_token":4.2e-07,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":true,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"gpt-oss-20b is a 20B-parameter open-weight model optimized for fast, efficient inference. It delivers strong reasoning and tool-use performance with low latency and memory footprint, making it ideal for local, edge, and cost-sensitive deployments","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"mistral-small-4-119b","providers":["hosted_vllm"],"max_tokens":120000.0,"quantization":null,"max_input_tokens":60000.0,"max_output_tokens":60000.0,"input_cost_per_token":5e-07,"output_cost_per_token":2.1e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Mistral Small 4 is a powerful hybrid model capable of acting as both a general instruction model and a reasoning model. It unifies the capabilities of three different model families—Instruct, Reasoning (previously called Magistral), and Devstral—into a single, unified model.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"Qwen-Image","providers":["hosted_vllm"],"max_tokens":null,"quantization":"FP16","max_input_tokens":null,"max_output_tokens":null,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":1.5e-08,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"image_generation","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":false,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"An updated Qwen text-to-image foundation model (2512) with enhanced human realism, finer natural details, and best-in-class complex text rendering and image editing.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"Qwen3-Embedding-8B","providers":["hosted_vllm"],"max_tokens":null,"quantization":null,"max_input_tokens":null,"max_output_tokens":null,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"embedding","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":false,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Qwen3 Embedding: advanced multilingual models for text embedding, ranking, retrieval, classification, and clustering.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":0.001,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"Qwen3-Reranker-4B","providers":["hosted_vllm"],"max_tokens":14000.0,"quantization":"FP16","max_input_tokens":14000.0,"max_output_tokens":14000.0,"input_cost_per_token":0.0,"output_cost_per_token":0.0,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"rerank","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":false,"supports_function_calling":false,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Qwen3 Reranker is a powerful multilingual 8 billion-parameter model designed for high-accuracy text ranking and retrieval across 100+ languages and long-form contexts","input_cost_per_second":null,"input_cost_per_query":0.01,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"qwen3.5-122b","providers":["hosted_vllm"],"max_tokens":240000.0,"quantization":"FP8","max_input_tokens":120000.0,"max_output_tokens":120000.0,"input_cost_per_token":1e-06,"output_cost_per_token":4.2e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":true,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Qwen3.5-122B-A10B-FP8 is a 122B-parameter Mixture-of-Experts vision-language model activating only 10B parameters per token, featuring hybrid Gated DeltaNet attention, 256 experts, 262K context, and 201 language support. FP8 quantized.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"qwen3.5-9b","providers":["hosted_vllm"],"max_tokens":200000.0,"quantization":"FP16","max_input_tokens":80000.0,"max_output_tokens":120000.0,"input_cost_per_token":7e-08,"output_cost_per_token":3.5e-07,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":false,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":false,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Qwen3.5 is the latest generation of large language models in Qwen series.","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null},{"model_group":"qwen3.8-27b","providers":["hosted_vllm"],"max_tokens":240000.0,"quantization":"NVFP4","max_input_tokens":120000.0,"max_output_tokens":120000.0,"input_cost_per_token":5e-07,"output_cost_per_token":2.1e-06,"input_cost_per_pixel":null,"ocr_cost_per_page":null,"input_cost_per_token_batches":null,"output_cost_per_token_batches":null,"mode":"chat","tpm":null,"rpm":null,"supports_parallel_function_calling":false,"supports_vision":true,"supports_web_search":false,"supports_url_context":false,"supports_reasoning":true,"supports_function_calling":true,"supports_audio_input":null,"supports_video_understanding":true,"supported_openai_params":["frequency_penalty","logit_bias","logprobs","top_logprobs","max_tokens","max_completion_tokens","modalities","prediction","n","presence_penalty","seed","stop","stream","stream_options","temperature","top_p","tools","tool_choice","function_call","functions","max_retries","extra_headers","parallel_tool_calls","audio","web_search_options","service_tier","safety_identifier","prompt_cache_key","prompt_cache_retention","store","response_format","reasoning_effort","thinking"],"configurable_clientside_auth_params":null,"description":"Qwen3.6-27b is a 27B-parameter, vision-language model optimized for agentic coding and repository level reasoning","input_cost_per_second":null,"input_cost_per_query":null,"input_cost_per_request":null,"usage_multiplier":null,"is_public_model_group":false,"health_status":null,"health_response_time":null,"health_checked_at":null}]}