{"data":[{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","_provider":"anthropic","pricing":{"prompt":"0.000003","completion":"0.000015"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","_provider":"anthropic","pricing":{"prompt":"0.000005","completion":"0.000025"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","_provider":"anthropic","pricing":{"prompt":"0.000005","completion":"0.000025"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","_provider":"anthropic","pricing":{"prompt":"0.0000001","completion":"0.0000005"},"context_length":64000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true,"max_completion_tokens":64000,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"]},{"id":"gpt-5","name":"gpt-5","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.00001"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-mini","name":"gpt-5-mini","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000002"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-mini","name":"gpt-4o-mini","_provider":"openai","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"max_completion_tokens":16384,"freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o1-pro","name":"o1-pro","_provider":"openai","pricing":{"prompt":"0.00015","completion":"0.0006"},"context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","_provider":"google","pricing":{"prompt":"0.000002","completion":"0.000012"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","_provider":"google","pricing":{"prompt":"0.00000125","completion":"0.00001"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","_provider":"google","pricing":{"prompt":"0.0000003","completion":"0.0000025"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65535,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-chat","name":"DeepSeek: DeepSeek V3","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","_provider":"deepseek","pricing":{"prompt":"0.0000002574","completion":"0.0000010287"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":16000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"groq/llama-3.1-8b-instant","name":"llama-3.1-8b-instant","_provider":"groq","pricing":{"prompt":"0.000000005","completion":"0.000000008"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"groq","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"mistral-large-latest","name":"mistral-large-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"chatgpt-image-latest","name":"chatgpt-image-latest","_provider":"openai","pricing":{"prompt":"0.0000005","completion":"0.000001"},"context_length":16384,"architecture":{"modality":"image","input_modalities":["text"],"output_modalities":["image"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo","name":"gpt-3.5-turbo","_provider":"openai","pricing":{"prompt":"0.0000005","completion":"0.0000015"},"context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":4096,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo-0125","name":"gpt-3.5-turbo-0125","_provider":"openai","pricing":{"prompt":"0.00000005","completion":"0.00000015"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo-1106","name":"gpt-3.5-turbo-1106","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000002"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo-16k","name":"gpt-3.5-turbo-16k","_provider":"openai","pricing":{"prompt":"0.000003","completion":"0.000004"},"context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":4096,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo-instruct","name":"gpt-3.5-turbo-instruct","_provider":"openai","pricing":{"prompt":"0.0000015","completion":"0.000002"},"context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing_source":"openrouter","description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"max_completion_tokens":4096,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-3.5-turbo-instruct-0914","name":"gpt-3.5-turbo-instruct-0914","_provider":"openai","pricing":{"prompt":"0.00000015","completion":"0.0000002"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4","name":"gpt-4","_provider":"openai","pricing":{"prompt":"0.00003","completion":"0.00006"},"context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":4096,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4-0613","name":"gpt-4-0613","_provider":"openai","pricing":{"prompt":"0.000003","completion":"0.000006"},"context_length":128000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4-turbo","name":"gpt-4-turbo","_provider":"openai","pricing":{"prompt":"0.00001","completion":"0.00003"},"context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":4096,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4.1","name":"gpt-4.1","_provider":"openai","pricing":{"prompt":"0.000002","completion":"0.000008"},"context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","_provider":"openai","pricing":{"prompt":"0.0000004","completion":"0.0000016"},"context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000004"},"context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o","name":"gpt-4o","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.00001"},"context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"max_completion_tokens":16384,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-mini-search-preview","name":"gpt-4o-mini-search-preview","_provider":"openai","pricing":{"prompt":"0.000000015","completion":"0.00000006"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-mini-transcribe","name":"gpt-4o-mini-transcribe","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.0000005"},"context_length":128000,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_transcriptions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-mini-tts","name":"gpt-4o-mini-tts","_provider":"openai","pricing":{"prompt":"0.00000006","completion":"0"},"context_length":128000,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_speech":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-search-preview","name":"gpt-4o-search-preview","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-transcribe","name":"gpt-4o-transcribe","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_transcriptions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-4o-transcribe-diarize","name":"gpt-4o-transcribe-diarize","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_transcriptions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-chat-latest","name":"gpt-5-chat-latest","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-codex","name":"gpt-5-codex","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-nano","name":"gpt-5-nano","_provider":"openai","pricing":{"prompt":"0.00000005","completion":"0.0000004"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-pro","name":"gpt-5-pro","_provider":"openai","pricing":{"prompt":"0.000015","completion":"0.00012"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5-search-api","name":"gpt-5-search-api","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.1","name":"gpt-5.1","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.00001"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.1-chat-latest","name":"gpt-5.1-chat-latest","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.000001"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.1-codex","name":"gpt-5.1-codex","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.00001"},"context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.1-codex-max","name":"gpt-5.1-codex-max","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.00001"},"context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.1-codex-mini","name":"gpt-5.1-codex-mini","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000002"},"context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.2","name":"gpt-5.2","_provider":"openai","pricing":{"prompt":"0.00000175","completion":"0.000014"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.2-chat-latest","name":"gpt-5.2-chat-latest","_provider":"openai","pricing":{"prompt":"0.000000175","completion":"0.0000014"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.2-codex","name":"gpt-5.2-codex","_provider":"openai","pricing":{"prompt":"0.00000175","completion":"0.000014"},"context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.2-pro","name":"gpt-5.2-pro","_provider":"openai","pricing":{"prompt":"0.000021","completion":"0.000168"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.4","name":"gpt-5.4","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.000015"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.4-mini","name":"gpt-5.4-mini","_provider":"openai","pricing":{"prompt":"0.00000075","completion":"0.0000045"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.4-nano","name":"gpt-5.4-nano","_provider":"openai","pricing":{"prompt":"0.0000002","completion":"0.00000125"},"context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.4-pro","name":"gpt-5.4-pro","_provider":"openai","pricing":{"prompt":"0.00003","completion":"0.00018"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.5","name":"gpt-5.5","_provider":"openai","pricing":{"prompt":"0.000005","completion":"0.00003"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.5-pro","name":"gpt-5.5-pro","_provider":"openai","pricing":{"prompt":"0.00003","completion":"0.00018"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.6-luna","name":"gpt-5.6-luna","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000006"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.6-sol","name":"gpt-5.6-sol","_provider":"openai","pricing":{"prompt":"0.000005","completion":"0.00003"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-5.6-terra","name":"gpt-5.6-terra","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000006"},"context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-audio","name":"gpt-audio","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.00001"},"context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":16384,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-audio-1.5","name":"gpt-audio-1.5","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.000001"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-audio-mini","name":"gpt-audio-mini","_provider":"openai","pricing":{"prompt":"0.0000006","completion":"0.0000024"},"context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"max_completion_tokens":16384,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-image-1","name":"gpt-image-1","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"image","input_modalities":["text"],"output_modalities":["image"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"images_generations":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-image-1-mini","name":"gpt-image-1-mini","_provider":"openai","pricing":{"prompt":"0.0000002","completion":"0"},"context_length":16384,"architecture":{"modality":"image","input_modalities":["text"],"output_modalities":["image"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-image-1.5","name":"gpt-image-1.5","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"image","input_modalities":["text"],"output_modalities":["image"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"images_generations":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-image-2","name":"gpt-image-2","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"image","input_modalities":["text"],"output_modalities":["image"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"images_generations":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-live-transcribe","name":"gpt-live-transcribe","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_transcriptions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-realtime","name":"gpt-realtime","_provider":"openai","pricing":{"prompt":"0.0000004","completion":"0.0000016"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-realtime-1.5","name":"gpt-realtime-1.5","_provider":"openai","pricing":{"prompt":"0.0000004","completion":"0.0000016"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-realtime-2","name":"gpt-realtime-2","_provider":"openai","pricing":{"prompt":"0.0000004","completion":"0.0000024"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-realtime-2.1","name":"gpt-realtime-2.1","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-realtime-2.1-mini","name":"gpt-realtime-2.1-mini","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-realtime-mini","name":"gpt-realtime-mini","_provider":"openai","pricing":{"prompt":"0.00000006","completion":"0.00000024"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gpt-realtime-translate","name":"gpt-realtime-translate","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-realtime-whisper","name":"gpt-realtime-whisper","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_transcriptions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gpt-transcribe","name":"gpt-transcribe","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_transcriptions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"o1","name":"o1","_provider":"openai","pricing":{"prompt":"0.000015","completion":"0.00006"},"context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o3","name":"o3","_provider":"openai","pricing":{"prompt":"0.000002","completion":"0.000008"},"context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o3-deep-research","name":"o3-deep-research","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000004"},"context_length":200000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o3-mini","name":"o3-mini","_provider":"openai","pricing":{"prompt":"0.0000011","completion":"0.0000044"},"context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o3-pro","name":"o3-pro","_provider":"openai","pricing":{"prompt":"0.00002","completion":"0.00008"},"context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o4-mini","name":"o4-mini","_provider":"openai","pricing":{"prompt":"0.0000011","completion":"0.0000044"},"context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing_source":"openrouter","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"max_completion_tokens":100000,"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"o4-mini-deep-research","name":"o4-mini-deep-research","_provider":"openai","pricing":{"prompt":"0.0000002","completion":"0.0000008"},"context_length":16384,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-chat-latest","name":"OpenAI: GPT Chat Latest","description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","_provider":"openai","pricing":{"prompt":"0.000005","completion":"0.00003"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-3.5-turbo:batch","name":"OpenAI: GPT-3.5 Turbo (batch)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","_provider":"openai","pricing":{"prompt":"0.00000025","completion":"0.00000075"},"pricing_source":"openrouter","context_length":16385,"max_completion_tokens":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000002"},"pricing_source":"openrouter","context_length":4095,"max_completion_tokens":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4-turbo:batch","name":"OpenAI: GPT-4 Turbo (batch)","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","_provider":"openai","pricing":{"prompt":"0.000005","completion":"0.000015"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":4096,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","_provider":"openai","pricing":{"prompt":"0.00001","completion":"0.00003"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4.1:batch","name":"OpenAI: GPT-4.1 (batch)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000004"},"pricing_source":"openrouter","context_length":1047576,"max_completion_tokens":32768,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4.1-mini:batch","name":"OpenAI: GPT-4.1 Mini (batch)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","_provider":"openai","pricing":{"prompt":"0.0000002","completion":"0.0000008"},"pricing_source":"openrouter","context_length":1047576,"max_completion_tokens":32768,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4.1-nano:batch","name":"OpenAI: GPT-4.1 Nano (batch)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","_provider":"openai","pricing":{"prompt":"0.00000005","completion":"0.0000002"},"pricing_source":"openrouter","context_length":1047576,"max_completion_tokens":32768,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4o:batch","name":"OpenAI: GPT-4o (batch)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.000005"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":16384,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-4o-mini:batch","name":"OpenAI: GPT-4o-mini (batch)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","_provider":"openai","pricing":{"prompt":"0.000000075","completion":"0.0000003"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":16384,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5:batch","name":"OpenAI: GPT-5 (batch)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","_provider":"openai","pricing":{"prompt":"0.000000625","completion":"0.000005"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5-codex:batch","name":"OpenAI: GPT-5 Codex (batch)","description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","_provider":"openai","pricing":{"prompt":"0.000000625","completion":"0.000005"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5-mini:batch","name":"OpenAI: GPT-5 Mini (batch)","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","_provider":"openai","pricing":{"prompt":"0.000000125","completion":"0.000001"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5-nano:batch","name":"OpenAI: GPT-5 Nano (batch)","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","_provider":"openai","pricing":{"prompt":"0.000000025","completion":"0.0000002"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5-pro:batch","name":"OpenAI: GPT-5 Pro (batch)","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","_provider":"openai","pricing":{"prompt":"0.0000075","completion":"0.00006"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.1:batch","name":"OpenAI: GPT-5.1 (batch)","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","_provider":"openai","pricing":{"prompt":"0.000000625","completion":"0.000005"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.2:batch","name":"OpenAI: GPT-5.2 (batch)","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","_provider":"openai","pricing":{"prompt":"0.000000875","completion":"0.000007"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","_provider":"openai","pricing":{"prompt":"0.00000175","completion":"0.000014"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":16384,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.2-pro:batch","name":"OpenAI: GPT-5.2 Pro (batch)","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","_provider":"openai","pricing":{"prompt":"0.0000105","completion":"0.000084"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.4:batch","name":"OpenAI: GPT-5.4 (batch)","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","_provider":"openai","pricing":{"prompt":"0.00000125","completion":"0.0000075"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","_provider":"openai","pricing":{"prompt":"0.000008","completion":"0.000015"},"pricing_source":"openrouter","context_length":272000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","top_logprobs"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.4-mini:batch","name":"OpenAI: GPT-5.4 Mini (batch)","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","_provider":"openai","pricing":{"prompt":"0.000000375","completion":"0.00000225"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.4-nano:batch","name":"OpenAI: GPT-5.4 Nano (batch)","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.000000625"},"pricing_source":"openrouter","context_length":400000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.4-pro:batch","name":"OpenAI: GPT-5.4 Pro (batch)","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","_provider":"openai","pricing":{"prompt":"0.000015","completion":"0.00009"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.5:batch","name":"OpenAI: GPT-5.5 (batch)","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.000015"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.5-pro:batch","name":"OpenAI: GPT-5.5 Pro (batch)","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","_provider":"openai","pricing":{"prompt":"0.000015","completion":"0.00009"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-luna:batch","name":"OpenAI: GPT-5.6 Luna (batch)","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-luna-pro:batch","name":"OpenAI: GPT-5.6 Luna Pro (batch)","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.0000001","completion":"0.0000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-sol:batch","name":"OpenAI: GPT-5.6 Sol (batch)","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.000015"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.000005","completion":"0.00003"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-sol-pro:batch","name":"OpenAI: GPT-5.6 Sol Pro (batch)","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.0000025","completion":"0.000015"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-terra:batch","name":"OpenAI: GPT-5.6 Terra (batch)","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-5.6-terra-pro:batch","name":"OpenAI: GPT-5.6 Terra Pro (batch)","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000006"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","_provider":"openai","pricing":{"prompt":"0.000000037","completion":"0.00000017"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","_provider":"openai","pricing":{"prompt":"0.00000003","completion":"0.00000013"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-oss-20b:free","name":"OpenAI: gpt-oss-20b (free)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","_provider":"openai","pricing":{"prompt":"0.000000075","completion":"0.0000003"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o1:batch","name":"OpenAI: o1 (batch)","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","_provider":"openai","pricing":{"prompt":"0.0000075","completion":"0.00003"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o1-pro:batch","name":"OpenAI: o1-pro (batch)","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","_provider":"openai","pricing":{"prompt":"0.000075","completion":"0.0003"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o3:batch","name":"OpenAI: o3 (batch)","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","_provider":"openai","pricing":{"prompt":"0.000001","completion":"0.000004"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o3-mini:batch","name":"OpenAI: o3 Mini (batch)","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","_provider":"openai","pricing":{"prompt":"0.00000055","completion":"0.0000022"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o3-mini-high","name":"OpenAI: o3 Mini High","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","_provider":"openai","pricing":{"prompt":"0.0000011","completion":"0.0000044"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o3-mini-high:batch","name":"OpenAI: o3 Mini High (batch)","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","_provider":"openai","pricing":{"prompt":"0.00000055","completion":"0.0000022"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o3-pro:batch","name":"OpenAI: o3 Pro (batch)","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","_provider":"openai","pricing":{"prompt":"0.00001","completion":"0.00004"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o4-mini:batch","name":"OpenAI: o4 Mini (batch)","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","_provider":"openai","pricing":{"prompt":"0.00000055","completion":"0.0000022"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o4-mini-high","name":"OpenAI: o4 Mini High","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","_provider":"openai","pricing":{"prompt":"0.0000011","completion":"0.0000044"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"openai/o4-mini-high:batch","name":"OpenAI: o4 Mini High (batch)","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","_provider":"openai","pricing":{"prompt":"0.00000055","completion":"0.0000022"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":100000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"text-embedding-3-large","name":"text-embedding-3-large","_provider":"openai","pricing":{"prompt":"0.000000013","completion":"0"},"context_length":16384,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"embeddings":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"text-embedding-3-small","name":"text-embedding-3-small","_provider":"openai","pricing":{"prompt":"0.000000002","completion":"0"},"context_length":16384,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"embeddings":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","_provider":"openai","pricing":{"prompt":"0.00000001","completion":"0"},"context_length":16384,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"embeddings":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"tts-1","name":"tts-1","_provider":"openai","pricing":{"prompt":"0.0000015","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_speech":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"tts-1-1106","name":"tts-1-1106","_provider":"openai","pricing":{"prompt":"0.0000015","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_speech":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"tts-1-hd","name":"tts-1-hd","_provider":"openai","pricing":{"prompt":"0.000003","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_speech":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"tts-1-hd-1106","name":"tts-1-hd-1106","_provider":"openai","pricing":{"prompt":"0.000003","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"openai","mode":"system","endpoints":{"audio_speech":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"whisper-1","name":"whisper-1","_provider":"openai","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"openai","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_transcriptions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"anthropic/claude-3-haiku","name":"Anthropic: Claude 3 Haiku","description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","_provider":"anthropic","pricing":{"prompt":"0.00000025","completion":"0.00000125"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":4096,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-fable-5:batch","name":"Anthropic: Claude Fable 5 (batch)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","_provider":"anthropic","pricing":{"prompt":"0.000005","completion":"0.000025"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-haiku-4-5:batch","name":"Anthropic: Claude Haiku 4.5 (batch)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","_provider":"anthropic","pricing":{"prompt":"0.0000005","completion":"0.0000025"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":64000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4","name":"Anthropic: Claude Opus 4","description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","_provider":"anthropic","pricing":{"prompt":"0.000015","completion":"0.000075"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":32000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-1","name":"Anthropic: Claude Opus 4.1","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","_provider":"anthropic","pricing":{"prompt":"0.000015","completion":"0.000075"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":32000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-1:batch","name":"Anthropic: Claude Opus 4.1 (batch)","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","_provider":"anthropic","pricing":{"prompt":"0.0000075","completion":"0.0000375"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":32000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-5:batch","name":"Anthropic: Claude Opus 4.5 (batch)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","_provider":"anthropic","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":64000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-6:batch","name":"Anthropic: Claude Opus 4.6 (batch)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","_provider":"anthropic","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-7:batch","name":"Anthropic: Claude Opus 4.7 (batch)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","_provider":"anthropic","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","_provider":"anthropic","pricing":{"prompt":"0.00003","completion":"0.00015"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-8:batch","name":"Anthropic: Claude Opus 4.8 (batch)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","_provider":"anthropic","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-4-8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","_provider":"anthropic","pricing":{"prompt":"0.00001","completion":"0.00005"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","_provider":"anthropic","pricing":{"prompt":"0.000003","completion":"0.000015"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":64000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-sonnet-4-5:batch","name":"Anthropic: Claude Sonnet 4.5 (batch)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","_provider":"anthropic","pricing":{"prompt":"0.0000015","completion":"0.0000075"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":64000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-sonnet-4-6:batch","name":"Anthropic: Claude Sonnet 4.6 (batch)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","_provider":"anthropic","pricing":{"prompt":"0.0000015","completion":"0.0000075"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-sonnet-5:batch","name":"Anthropic: Claude Sonnet 5 (batch)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","_provider":"anthropic","pricing":{"prompt":"0.000001","completion":"0.000005"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","_provider":"anthropic","pricing":{"prompt":"0.00001","completion":"0.00005"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","_provider":"anthropic","pricing":{"prompt":"0.0000005","completion":"0.0000025"},"context_length":64000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true,"max_completion_tokens":64000,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"]},{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","_provider":"anthropic","pricing":{"prompt":"0.000005","completion":"0.000025"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-opus-5","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","_provider":"anthropic","pricing":{"prompt":"0.000005","completion":"0.000025"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-5:batch","name":"Claude Opus 5 (batch)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","_provider":"anthropic","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"anthropic/claude-opus-5-fast","name":"Claude Opus 5 (Fast)","description":"Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","_provider":"anthropic","pricing":{"prompt":"0.00001","completion":"0.00005"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","_provider":"anthropic","pricing":{"prompt":"0.0000003","completion":"0.0000015"},"context_length":64000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true,"max_completion_tokens":64000,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"]},{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","_provider":"anthropic","pricing":{"prompt":"0.000002","completion":"0.00001"},"context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing_source":"openrouter","supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"anthropic","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"antigravity-preview-05-2026","name":"Antigravity Agent Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"deep-research-max-preview-04-2026","name":"Deep Research Max Preview (Apr-21-2026)","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"deep-research-preview-04-2026","name":"Deep Research Preview (Apr-21-2026)","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"deep-research-pro-preview-12-2025","name":"Deep Research Pro Preview (Dec-12-2025)","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-2.0-flash","name":"Gemini 2.0 Flash","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-2.0-flash-001","name":"Gemini 2.0 Flash 001","_provider":"google","pricing":{"prompt":"0.00000001","completion":"0.00000004"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.0-flash-lite","name":"Gemini 2.0 Flash-Lite","_provider":"google","pricing":{"prompt":"0.0000000075","completion":"0.00000003"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.0-flash-lite-001","name":"Gemini 2.0 Flash-Lite 001","_provider":"google","pricing":{"prompt":"0.0000000075","completion":"0.00000003"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview 10-2025","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-2.5-flash-preview-tts","name":"Gemini 2.5 Flash Preview TTS","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","_provider":"google","pricing":{"prompt":"0.0000001","completion":"0.0000004"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65535,"freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.5-pro-preview-tts","name":"Gemini 2.5 Pro Preview TTS","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","_provider":"google","pricing":{"prompt":"0.0000005","completion":"0.000003"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3-pro-preview","name":"Gemini 3 Pro Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","_provider":"google","pricing":{"prompt":"0.00000025","completion":"0.0000015"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","_provider":"google","pricing":{"prompt":"0.00000025","completion":"0.0000015"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","_provider":"google","pricing":{"prompt":"0.000002","completion":"0.000012"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","_provider":"google","pricing":{"prompt":"0.0000015","completion":"0.000009"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","_provider":"google","pricing":{"prompt":"0.0000003","completion":"0.0000025"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","_provider":"google","pricing":{"prompt":"0.0000015","completion":"0.0000075"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-flash-latest","name":"Gemini Flash Latest","_provider":"google","pricing":{"prompt":"0.0000015","completion":"0.0000075"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing_source":"openrouter","description":"This model always redirects to the latest model in the Google Gemini Flash family.","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-pro-latest","name":"Gemini Pro Latest","_provider":"google","pricing":{"prompt":"0.000002","completion":"0.000012"},"context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing_source":"openrouter","description":"This model always redirects to the latest model in the Google Gemini Pro family.","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-robotics-er-1.5-preview","name":"Gemini Robotics-ER 1.5 Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-robotics-er-1.6-preview","name":"Gemini Robotics-ER 1.6 Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"gemini-robotics-er-2-preview","name":"Gemini Robotics-ER 2 Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"google/gemini-2.5-flash:batch","name":"Google: Gemini 2.5 Flash (batch)","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","_provider":"google","pricing":{"prompt":"0.00000015","completion":"0.00000125"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65535,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-2.5-flash-lite:batch","name":"Google: Gemini 2.5 Flash Lite (batch)","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","_provider":"google","pricing":{"prompt":"0.00000005","completion":"0.0000002"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65535,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-2.5-pro:batch","name":"Google: Gemini 2.5 Pro (batch)","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","_provider":"google","pricing":{"prompt":"0.000000625","completion":"0.000005"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","_provider":"google","pricing":{"prompt":"0.00000125","completion":"0.00001"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65535,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","_provider":"google","pricing":{"prompt":"0.00000125","completion":"0.00001"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3-flash-preview:batch","name":"Google: Gemini 3 Flash Preview (batch)","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","_provider":"google","pricing":{"prompt":"0.00000025","completion":"0.0000015"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3.1-flash-lite:batch","name":"Google: Gemini 3.1 Flash Lite (batch)","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","_provider":"google","pricing":{"prompt":"0.000000125","completion":"0.00000075"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3.1-pro-preview:batch","name":"Google: Gemini 3.1 Pro Preview (batch)","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","_provider":"google","pricing":{"prompt":"0.000001","completion":"0.000006"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3.5-flash:batch","name":"Google: Gemini 3.5 Flash (batch)","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","_provider":"google","pricing":{"prompt":"0.00000075","completion":"0.0000045"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3.5-flash-lite:batch","name":"Google: Gemini 3.5 Flash Lite (batch)","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","_provider":"google","pricing":{"prompt":"0.00000015","completion":"0.00000125"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemini-3.6-flash:batch","name":"Google: Gemini 3.6 Flash (batch)","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","_provider":"google","pricing":{"prompt":"0.00000075","completion":"0.00000375"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":65536,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","_provider":"google","pricing":{"prompt":"0.00000065","completion":"0.00000065"},"pricing_source":"openrouter","context_length":8192,"max_completion_tokens":2048,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemma-3-27b-it","name":"Google: Gemma 3 27B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","_provider":"google","pricing":{"prompt":"0.00000008","completion":"0.00000045"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","_provider":"google","pricing":{"prompt":"0.00000006","completion":"0.00000012"},"pricing_source":"openrouter","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/codegemma-1.1-7b","name":"google/codegemma-1.1-7b","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"google/codegemma-7b","name":"google/codegemma-7b","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"google/diffusiongemma-26b-a4b-it","name":"google/diffusiongemma-26b-a4b-it","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"google/gemma-2b","name":"google/gemma-2b","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"google/gemma-3-12b-it","name":"google/gemma-3-12b-it","_provider":"google","pricing":{"prompt":"0.00000005","completion":"0.00000015"},"context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing_source":"openrouter","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":16384,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/gemma-3-4b-it","name":"google/gemma-3-4b-it","_provider":"google","pricing":{"prompt":"0.00000005","completion":"0.0000001"},"context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing_source":"openrouter","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"max_completion_tokens":16384,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"google/recurrentgemma-2b","name":"google/recurrentgemma-2b","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-2.5-flash-image","name":"Nano Banana","_provider":"google","pricing":{"prompt":"0.0000003","completion":"0.0000025"},"context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"max_completion_tokens":8192,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2","_provider":"google","pricing":{"prompt":"0.0000005","completion":"0.000003"},"context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","_provider":"google","pricing":{"prompt":"0.0000005","completion":"0.000003"},"context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","_provider":"google","pricing":{"prompt":"0.00000025","completion":"0.0000015"},"context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","temperature","top_p"],"max_completion_tokens":65536,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro","_provider":"google","pricing":{"prompt":"0.000002","completion":"0.000012"},"context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"gemini-3-pro-image","name":"Nano Banana Pro","_provider":"google","pricing":{"prompt":"0.000002","completion":"0.000012"},"context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing_source":"openrouter","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"max_completion_tokens":32768,"routeability":{"status":"routeable","routeable":true,"provider":"google","mode":"system","endpoints":{"images_generations":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nano-banana-pro-preview","name":"Nano Banana Pro","_provider":"google","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"google","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"codestral-2508","name":"codestral-2508","_provider":"mistral","pricing":{"prompt":"0.0000003","completion":"0.0000009"},"context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","supported_parameters":["frequency_penalty","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"codestral-embed","name":"codestral-embed","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"codestral-embed-2505","name":"codestral-embed-2505","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"codestral-latest","name":"codestral-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":256000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"devstral-2512","name":"devstral-2512","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"devstral-latest","name":"devstral-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"devstral-medium-latest","name":"devstral-medium-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"labs-leanstral-1-5","name":"labs-leanstral-1-5","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"labs-leanstral-1-5-1","name":"labs-leanstral-1-5-1","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"magistral-small-latest","name":"magistral-small-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"ministral-14b-2512","name":"ministral-14b-2512","_provider":"mistral","pricing":{"prompt":"0.0000002","completion":"0.0000002"},"context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"ministral-14b-latest","name":"ministral-14b-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"ministral-3b-2512","name":"ministral-3b-2512","_provider":"mistral","pricing":{"prompt":"0.0000001","completion":"0.0000001"},"context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"ministral-3b-latest","name":"ministral-3b-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"ministral-8b-2512","name":"ministral-8b-2512","_provider":"mistral","pricing":{"prompt":"0.00000015","completion":"0.00000015"},"context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"ministral-8b-latest","name":"ministral-8b-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","_provider":"mistralai","pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_source":"openrouter","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistral-code-agent-latest","name":"mistral-code-agent-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-code-fim-latest","name":"mistral-code-fim-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":256000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-code-latest","name":"mistral-code-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":256000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-embed","name":"mistral-embed","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-embed-2312","name":"mistral-embed-2312","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":8192,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-large-2512","name":"mistral-large-2512","_provider":"mistral","pricing":{"prompt":"0.0000005","completion":"0.0000015"},"context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"mistral-medium","name":"mistral-medium","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-medium-2505","name":"mistral-medium-2505","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-medium-2508","name":"mistral-medium-2508","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-medium-2604","name":"mistral-medium-2604","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-medium-3","name":"mistral-medium-3","_provider":"mistral","pricing":{"prompt":"0.0000004","completion":"0.000002"},"context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"mistral-medium-3-5","name":"mistral-medium-3-5","_provider":"mistral","pricing":{"prompt":"0.0000015","completion":"0.0000075"},"context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"mistral-medium-latest","name":"mistral-medium-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-moderation-2603","name":"mistral-moderation-2603","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-2512","name":"mistral-ocr-2512","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-3","name":"mistral-ocr-3","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-3-0","name":"mistral-ocr-3-0","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-4","name":"mistral-ocr-4","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-4-0","name":"mistral-ocr-4-0","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-4-1","name":"mistral-ocr-4-1","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-ocr-latest","name":"mistral-ocr-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-small-2603","name":"mistral-small-2603","_provider":"mistral","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"mistral","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"mistral-small-latest","name":"mistral-small-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-vibe-cli-fast","name":"mistral-vibe-cli-fast","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-vibe-cli-latest","name":"mistral-vibe-cli-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistral-vibe-cli-with-tools","name":"mistral-vibe-cli-with-tools","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"mistralai/mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","_provider":"mistralai","pricing":{"prompt":"0.0000004","completion":"0.000002"},"pricing_source":"openrouter","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-nemo","name":"Mistral: Mistral Nemo","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","_provider":"mistralai","pricing":{"prompt":"0.000000019","completion":"0.00000003"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","_provider":"mistralai","pricing":{"prompt":"0.00000005","completion":"0.00000008"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","_provider":"mistralai","pricing":{"prompt":"0.000000351","completion":"0.000000555"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","_provider":"mistralai","pricing":{"prompt":"0.00000009375","completion":"0.00000025"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","_provider":"mistralai","pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_source":"openrouter","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-saba","name":"Mistral: Saba","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","_provider":"mistralai","pricing":{"prompt":"0.0000002","completion":"0.0000006"},"pricing_source":"openrouter","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","_provider":"mistralai","pricing":{"prompt":"0.0000001","completion":"0.0000003"},"pricing_source":"openrouter","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/codestral-22b-instruct-v0.1","name":"mistralai/codestral-22b-instruct-v0.1","_provider":"mistralai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-7b-instruct-v0.3","name":"mistralai/mistral-7b-instruct-v0.3","_provider":"mistralai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-large","name":"mistralai/mistral-large","_provider":"mistralai","pricing":{"prompt":"0.000002","completion":"0.000006"},"context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing_source":"openrouter","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-large-2-instruct","name":"mistralai/mistral-large-2-instruct","_provider":"mistralai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mistral-nemotron","name":"mistralai/mistral-nemotron","_provider":"mistralai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mistralai/mixtral-8x22b-v0.1","name":"mistralai/mixtral-8x22b-v0.1","_provider":"mistralai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistralai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"voxtral-mini-2602","name":"voxtral-mini-2602","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-mini-latest","name":"voxtral-mini-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-mini-realtime-2602","name":"voxtral-mini-realtime-2602","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-mini-realtime-latest","name":"voxtral-mini-realtime-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-mini-transcribe-realtime-2602","name":"voxtral-mini-transcribe-realtime-2602","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"mistral","mode":"system","reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions","endpoints":{"audio_transcriptions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions"}}},"routeable":false},{"id":"voxtral-mini-tts-2603","name":"voxtral-mini-tts-2603","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":4096,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_speech":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-mini-tts-latest","name":"voxtral-mini-tts-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":4096,"architecture":{"modality":"audio","input_modalities":["text"],"output_modalities":["audio"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"audio_speech":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-small-2507","name":"voxtral-small-2507","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"voxtral-small-latest","name":"voxtral-small-latest","_provider":"mistral","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"mistral","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/allam-2-7b","name":"allam-2-7b","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":4096,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/canopylabs/orpheus-arabic-saudi","name":"canopylabs/orpheus-arabic-saudi","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":4000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/canopylabs/orpheus-v1-english","name":"canopylabs/orpheus-v1-english","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":4000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/compound","name":"groq/compound","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/compound-mini","name":"groq/compound-mini","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/llama-3.3-70b-versatile","name":"llama-3.3-70b-versatile","_provider":"groq","pricing":{"prompt":"0.000000059","completion":"0.000000079"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"groq","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"groq/meta-llama/llama-prompt-guard-2-22m","name":"meta-llama/llama-prompt-guard-2-22m","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":512,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/meta-llama/llama-prompt-guard-2-86m","name":"meta-llama/llama-prompt-guard-2-86m","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":512,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"groq","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"groq/openai/gpt-oss-120b","name":"openai/gpt-oss-120b","_provider":"groq","pricing":{"prompt":"0.000000015","completion":"0.00000006"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"groq","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"groq/openai/gpt-oss-20b","name":"openai/gpt-oss-20b","_provider":"groq","pricing":{"prompt":"0.0000000075","completion":"0.00000003"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"groq","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"groq/openai/gpt-oss-safeguard-20b","name":"openai/gpt-oss-safeguard-20b","_provider":"groq","pricing":{"prompt":"0.0000000075","completion":"0.00000003"},"context_length":131072,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"groq","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"groq/whisper-large-v3","name":"whisper-large-v3","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":448,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"groq","mode":"system","reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions","endpoints":{"audio_transcriptions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions"}}},"routeable":false},{"id":"groq/whisper-large-v3-turbo","name":"whisper-large-v3-turbo","_provider":"groq","pricing":{"prompt":"0","completion":"0"},"context_length":448,"architecture":{"modality":"audio","input_modalities":["audio"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"groq","mode":"system","reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions","endpoints":{"audio_transcriptions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into audio_transcriptions"}}},"routeable":false},{"id":"deepseek-ai/deepseek-coder-6.7b-instruct","name":"deepseek-ai/deepseek-coder-6.7b-instruct","_provider":"deepseek-ai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"deepseek","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"deepseek-ai/deepseek-v4-flash-0731","_provider":"deepseek-ai","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"deepseek","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","_provider":"deepseek","pricing":{"prompt":"0.00000014","completion":"0.00000028"},"context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing_source":"openrouter","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"max_completion_tokens":393216,"freeTier":true,"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","_provider":"deepseek","pricing":{"prompt":"0.00000027","completion":"0.00000112"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","_provider":"deepseek","pricing":{"prompt":"0.00000025","completion":"0.00000095"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","_provider":"deepseek","pricing":{"prompt":"0.00000027","completion":"0.000001"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","_provider":"deepseek","pricing":{"prompt":"0.000000269","completion":"0.0000004"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","_provider":"deepseek","pricing":{"prompt":"0.00000027","completion":"0.00000041"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","_provider":"deepseek","pricing":{"prompt":"0.00000009","completion":"0.00000018"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":384000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-r1","name":"DeepSeek: R1","description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","_provider":"deepseek","pricing":{"prompt":"0.0000007","completion":"0.0000025"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":16000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"supported_parameters":["frequency_penalty","include_reasoning","max_completion_tokens","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek: R1 0528","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","_provider":"deepseek","pricing":{"prompt":"0.0000005","completion":"0.00000215"},"pricing_source":"openrouter","context_length":163840,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","_provider":"deepseek","pricing":{"prompt":"0.0000008","completion":"0.0000008"},"pricing_source":"openrouter","context_length":8192,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"deepseek","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-4.5","name":"glm-4.5","_provider":"zai","pricing":{"prompt":"0.0000006","completion":"0.0000022"},"context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":98304,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-4.5-air","name":"glm-4.5-air","_provider":"zai","pricing":{"prompt":"0.00000013","completion":"0.00000085"},"context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":98304,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-5","name":"glm-5","_provider":"zai","pricing":{"prompt":"0.00000095","completion":"0.00000255"},"context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"max_completion_tokens":131072,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-5-turbo","name":"glm-5-turbo","_provider":"zai","pricing":{"prompt":"0.0000012","completion":"0.000004"},"context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":131072,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-5.1","name":"glm-5.1","_provider":"zai","pricing":{"prompt":"0.000000952","completion":"0.000002992"},"context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"max_completion_tokens":131072,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"glm-5.2","name":"glm-5.2","_provider":"zai","pricing":{"prompt":"0.000000098","completion":"0.000000308"},"context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"max_completion_tokens":128000,"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"fireworks/deepseek-v4-flash","name":"accounts/fireworks/models/deepseek-v4-flash","_provider":"fireworks","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"fireworks","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"fireworks/deepseek-v4-flash-0731","name":"accounts/fireworks/models/deepseek-v4-flash-0731","_provider":"fireworks","pricing":{"prompt":"0","completion":"0"},"context_length":1048576,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"fireworks","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"fireworks/qwen3-embedding-8b","name":"accounts/fireworks/models/qwen3-embedding-8b","_provider":"fireworks","pricing":{"prompt":"0","completion":"0"},"context_length":40960,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"fireworks","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"fireworks/qwen3-reranker-8b","name":"accounts/fireworks/models/qwen3-reranker-8b","_provider":"fireworks","pricing":{"prompt":"0","completion":"0"},"context_length":40960,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"fireworks","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"fireworks/qwen3p7-plus","name":"accounts/fireworks/models/qwen3p7-plus","_provider":"fireworks","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"fireworks","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"c4ai-aya-expanse-32b","name":"c4ai-aya-expanse-32b","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":128000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"c4ai-aya-vision-32b","name":"c4ai-aya-vision-32b","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":16384,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"cohere/command-a","name":"Cohere: Command A","description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","_provider":"cohere","pricing":{"prompt":"0.0000025","completion":"0.00001"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"cohere","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"cohere/north-mini-code:free","name":"Cohere: North Mini Code (free)","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":64000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"cohere","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"command-a-03-2025","name":"command-a-03-2025","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":288000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"command-a-plus-05-2026","name":"command-a-plus-05-2026","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":436000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"command-a-reasoning-08-2025","name":"command-a-reasoning-08-2025","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":288768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"command-a-translate-08-2025","name":"command-a-translate-08-2025","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":8992,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"command-a-vision-07-2025","name":"command-a-vision-07-2025","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":128000,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"command-r-08-2024","name":"command-r-08-2024","_provider":"cohere","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing_source":"openrouter","description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":4000,"routeability":{"status":"routeable","routeable":true,"provider":"cohere","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"command-r-plus-08-2024","name":"command-r-plus-08-2024","_provider":"cohere","pricing":{"prompt":"0.0000025","completion":"0.00001"},"context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing_source":"openrouter","description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"max_completion_tokens":4000,"routeability":{"status":"routeable","routeable":true,"provider":"cohere","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"command-r7b-12-2024","name":"command-r7b-12-2024","_provider":"cohere","pricing":{"prompt":"0.0000000375","completion":"0.00000015"},"context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing_source":"openrouter","description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"max_completion_tokens":4000,"routeability":{"status":"routeable","routeable":true,"provider":"cohere","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"command-r7b-arabic-02-2025","name":"command-r7b-arabic-02-2025","_provider":"cohere","pricing":{"prompt":"0","completion":"0"},"context_length":128000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"cohere","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"meta-llama/llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","_provider":"meta-llama","pricing":{"prompt":"0.0000004","completion":"0.0000004"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","_provider":"meta-llama","pricing":{"prompt":"0.00000005","completion":"0.00000008"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","_provider":"meta-llama","pricing":{"prompt":"0.000000027","completion":"0.000000201"},"pricing_source":"openrouter","context_length":60000,"max_completion_tokens":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","_provider":"meta-llama","pricing":{"prompt":"0.00000005","completion":"0.00000033"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","_provider":"meta-llama","pricing":{"prompt":"0.0000001","completion":"0.00000032"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-4-maverick","name":"Meta: Llama 4 Maverick","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","_provider":"meta-llama","pricing":{"prompt":"0.0000002","completion":"0.0000008"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-4-scout","name":"Meta: Llama 4 Scout","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","_provider":"meta-llama","pricing":{"prompt":"0.0000001","completion":"0.0000003"},"pricing_source":"openrouter","context_length":1310720,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta-llama/llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","_provider":"meta-llama","pricing":{"prompt":"0.00000018","completion":"0.00000018"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta-llama","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/muse-spark-1.1","name":"Meta: Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","_provider":"meta","pricing":{"prompt":"0.00000125","completion":"0.00000425"},"pricing_source":"openrouter","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/muse-spark-1.2","name":"Meta: Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","_provider":"meta","pricing":{"prompt":"0.00000125","completion":"0.00000425"},"pricing_source":"openrouter","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/codellama-70b","name":"meta/codellama-70b","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.1-70b-instruct","name":"meta/llama-3.1-70b-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.1-8b-instruct","name":"meta/llama-3.1-8b-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.2-11b-vision-instruct","name":"meta/llama-3.2-11b-vision-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.2-1b-instruct","name":"meta/llama-3.2-1b-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.2-3b-instruct","name":"meta/llama-3.2-3b-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.2-90b-vision-instruct","name":"meta/llama-3.2-90b-vision-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-3.3-70b-instruct","name":"meta/llama-3.3-70b-instruct","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama-guard-4-12b","name":"meta/llama-guard-4-12b","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meta/llama2-70b","name":"meta/llama2-70b","_provider":"meta","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"unrouteable","routeable":false,"provider":"meta","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","_provider":"qwen","pricing":{"prompt":"0.0000004","completion":"0.0000012"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen-plus","name":"Qwen: Qwen-Plus","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","_provider":"qwen","pricing":{"prompt":"0.00000026","completion":"0.00000078"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","_provider":"qwen","pricing":{"prompt":"0.0000001","completion":"0.0000002"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","_provider":"qwen","pricing":{"prompt":"0.00000025","completion":"0.00000075"},"pricing_source":"openrouter","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-14b","name":"Qwen: Qwen3 14B","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","_provider":"qwen","pricing":{"prompt":"0.0000002275","completion":"0.00000091"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","_provider":"qwen","pricing":{"prompt":"0.000000455","completion":"0.00000182"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","_provider":"qwen","pricing":{"prompt":"0.00000009","completion":"0.00000055"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","_provider":"qwen","pricing":{"prompt":"0.00000023","completion":"0.0000023"},"pricing_source":"openrouter","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","_provider":"qwen","pricing":{"prompt":"0.00000012","completion":"0.0000005"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","_provider":"qwen","pricing":{"prompt":"0.00000004815","completion":"0.00000019305"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","_provider":"qwen","pricing":{"prompt":"0.0000002","completion":"0.0000024"},"pricing_source":"openrouter","context_length":81920,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-32b","name":"Qwen: Qwen3 32B","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","_provider":"qwen","pricing":{"prompt":"0.00000008","completion":"0.00000028"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-8b","name":"Qwen: Qwen3 8B","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","_provider":"qwen","pricing":{"prompt":"0.000000117","completion":"0.000000455"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","_provider":"qwen","pricing":{"prompt":"0.00000007","completion":"0.00000027"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","_provider":"qwen","pricing":{"prompt":"0.0000003","completion":"0.000001"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","_provider":"qwen","pricing":{"prompt":"0.000000195","completion":"0.000000975"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","_provider":"qwen","pricing":{"prompt":"0.00000012","completion":"0.0000008"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","_provider":"qwen","pricing":{"prompt":"0.00000065","completion":"0.00000325"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-max","name":"Qwen: Qwen3 Max","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","_provider":"qwen","pricing":{"prompt":"0.00000078","completion":"0.0000039"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","_provider":"qwen","pricing":{"prompt":"0.00000078","completion":"0.0000039"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","_provider":"qwen","pricing":{"prompt":"0.00000009","completion":"0.0000011"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","_provider":"qwen","pricing":{"prompt":"0.00000015","completion":"0.0000012"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","_provider":"qwen","pricing":{"prompt":"0.00000021","completion":"0.0000019"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","_provider":"qwen","pricing":{"prompt":"0.0000004","completion":"0.000004"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","_provider":"qwen","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","_provider":"qwen","pricing":{"prompt":"0.0000002","completion":"0.0000024"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","_provider":"qwen","pricing":{"prompt":"0.000000104","completion":"0.000000416"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","_provider":"qwen","pricing":{"prompt":"0.000000117","completion":"0.000000455"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","_provider":"qwen","pricing":{"prompt":"0.00000018","completion":"0.0000021"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3.7-flash","name":"Qwen: Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","_provider":"qwen","pricing":{"prompt":"0.00000003","completion":"0.00000013"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3.7-max","name":"Qwen: Qwen3.7 Max","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","_provider":"qwen","pricing":{"prompt":"0.000001475","completion":"0.000004425"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","_provider":"qwen","pricing":{"prompt":"0.00000032","completion":"0.00000128"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen3.8-max","name":"Qwen: Qwen3.8 Max","description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","_provider":"qwen","pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","_provider":"qwen","pricing":{"prompt":"0.00000036","completion":"0.0000004"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","_provider":"qwen","pricing":{"prompt":"0.00000066","completion":"0.000001"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"qwen","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"ai21/jamba-large-1.7-2025-07","name":"AI21: Jamba Large 1.7","description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","_provider":"ai21","pricing":{"prompt":"0.0000002","completion":"0.0000008"},"context_length":256000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"unrouteable","routeable":false,"provider":"ai21","mode":"system","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false,"max_completion_tokens":4096,"supported_parameters":["max_tokens","response_format","stop","temperature","tool_choice","tools","top_p"]},{"id":"ai21/jamba-mini-2-2026-01","name":"AI21: Jamba Mini 2","_provider":"ai21","pricing":{"prompt":"0.00000002","completion":"0.00000004"},"context_length":256000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"unrouteable","routeable":false,"provider":"ai21","mode":"system","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","_provider":"aion-labs","pricing":{"prompt":"0.0000008","completion":"0.0000016"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"aion-labs","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","_provider":"aion-labs","pricing":{"prompt":"0.000003","completion":"0.000006"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"aion-labs","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","_provider":"aion-labs","pricing":{"prompt":"0.0000007","completion":"0.0000014"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"aion-labs","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","_provider":"aion-labs","pricing":{"prompt":"0.0000008","completion":"0.0000016"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","temperature","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"aion-labs","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"allenai/olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","_provider":"allenai","pricing":{"prompt":"0.00000015","completion":"0.0000005"},"pricing_source":"openrouter","context_length":65536,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"allenai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"amazon/nova-2-lite-v1","name":"Amazon: Nova 2 Lite","description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","_provider":"amazon","pricing":{"prompt":"0.0000003","completion":"0.0000025"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":65535,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"amazon","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"amazon/nova-lite-v1","name":"Amazon: Nova Lite 1.0","description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","_provider":"amazon","pricing":{"prompt":"0.00000006","completion":"0.00000024"},"pricing_source":"openrouter","context_length":300000,"max_completion_tokens":5120,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"amazon","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"amazon/nova-micro-v1","name":"Amazon: Nova Micro 1.0","description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","_provider":"amazon","pricing":{"prompt":"0.000000035","completion":"0.00000014"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":5120,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"amazon","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"amazon/nova-premier-v1","name":"Amazon: Nova Premier 1.0","description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","_provider":"amazon","pricing":{"prompt":"0.0000025","completion":"0.0000125"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":32000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"amazon","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"amazon/nova-pro-v1","name":"Amazon: Nova Pro 1.0","description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","_provider":"amazon","pricing":{"prompt":"0.0000008","completion":"0.0000032"},"pricing_source":"openrouter","context_length":300000,"max_completion_tokens":5120,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"amazon","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"arcee-ai/trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","_provider":"arcee-ai","pricing":{"prompt":"0.00000022","completion":"0.00000085"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"arcee-ai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"arcee-ai/virtuoso-large","name":"Arcee AI: Virtuoso Large","description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","_provider":"arcee-ai","pricing":{"prompt":"0.00000075","completion":"0.0000012"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":64000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"arcee-ai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","_provider":"baidu","pricing":{"prompt":"0.00000042","completion":"0.00000125"},"pricing_source":"openrouter","context_length":123000,"max_completion_tokens":16000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"baidu","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"bytedance-seed/seed-1.6","name":"ByteDance Seed: Seed 1.6","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","_provider":"bytedance-seed","pricing":{"prompt":"0.00000025","completion":"0.000002"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"bytedance-seed","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"bytedance-seed/seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","_provider":"bytedance-seed","pricing":{"prompt":"0.000000075","completion":"0.0000003"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"bytedance-seed","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"bytedance-seed/seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","_provider":"bytedance-seed","pricing":{"prompt":"0.00000025","completion":"0.000002"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"bytedance-seed","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"bytedance-seed/seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","_provider":"bytedance-seed","pricing":{"prompt":"0.0000001","completion":"0.0000004"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"bytedance-seed","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"bytedance/ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","_provider":"bytedance","pricing":{"prompt":"0.0000001","completion":"0.0000002"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":2048,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"bytedance","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"deepcogito/cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","_provider":"deepcogito","pricing":{"prompt":"0.00000125","completion":"0.00000125"},"pricing_source":"openrouter","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"deepcogito","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sambanova/DeepSeek-V3.1","name":"DeepSeek-V3.1","_provider":"sambanova","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"sambanova","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"sambanova/DeepSeek-V3.2","name":"DeepSeek-V3.2","_provider":"sambanova","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"sambanova","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"sambanova/gpt-oss-120b","name":"gpt-oss-120b","_provider":"sambanova","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"sambanova","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","_provider":"ibm-granite","pricing":{"prompt":"0.000000017","completion":"0.000000112"},"pricing_source":"openrouter","context_length":131000,"max_completion_tokens":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"ibm-granite","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"ibm-granite/granite-4.1-8b","name":"IBM: Granite 4.1 8B","description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","_provider":"ibm-granite","pricing":{"prompt":"0.00000005","completion":"0.0000001"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"ibm-granite","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inception/mercury-2","name":"Inception: Mercury 2","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","_provider":"inception","pricing":{"prompt":"0.00000025","completion":"0.00000075"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":50000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inception","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inclusionai/ling-3.0-tiny:free","name":"inclusionAI: Ling 3.0 Tiny (free)","description":"Ling 3.0 Tiny is a mixture-of-experts model from InclusionAI, with 1.3B active parameters out of 7.9B total. It is designed for responsive agents, instruction following, and multi-turn conversations, with switchable...","_provider":"inclusionai","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inclusionai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inclusionai/ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","_provider":"inclusionai","pricing":{"prompt":"0.000000075","completion":"0.000000625"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inclusionai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inclusionai/ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","_provider":"inclusionai","pricing":{"prompt":"0.00000001","completion":"0.00000003"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inclusionai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inclusionai/ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","_provider":"inclusionai","pricing":{"prompt":"0.000000075","completion":"0.000000625"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inclusionai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"kimi-k2.7-code","name":"kimi-k2.7-code","_provider":"moonshot","pricing":{"prompt":"0.0000007","completion":"0.0000035"},"context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"max_completion_tokens":262144,"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"kimi-k2.7-code-highspeed","name":"kimi-k2.7-code-highspeed","_provider":"moonshot","pricing":{"prompt":"0.00000019","completion":"0.0000008"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"kimi-k3","name":"kimi-k3","_provider":"moonshot","pricing":{"prompt":"0.000003","completion":"0.000015"},"context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"kwaipilot/kat-coder-air-v2.5","name":"Kwaipilot: KAT-Coder-Air V2.5","description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","_provider":"kwaipilot","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":80000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"kwaipilot","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"kwaipilot/kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","_provider":"kwaipilot","pricing":{"prompt":"0.0000003","completion":"0.0000012"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":80000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"kwaipilot","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"kwaipilot/kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","_provider":"kwaipilot","pricing":{"prompt":"0.00000074","completion":"0.00000296"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":80000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"kwaipilot","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"inclusionai/ling-3.0-flash","name":"Ling-3.0-flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","_provider":"inclusionai","pricing":{"prompt":"0.000000021","completion":"0.000000063"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"inclusionai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","_provider":"anthracite-org","pricing":{"prompt":"0.000003","completion":"0.000005"},"pricing_source":"openrouter","context_length":16384,"max_completion_tokens":2048,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"anthracite-org","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"mancer/weaver","name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","_provider":"mancer","pricing":{"prompt":"0.0000005","completion":"0.00000075"},"pricing_source":"openrouter","context_length":8000,"max_completion_tokens":6000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"mancer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"meituan/longcat-2.0","name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","_provider":"meituan","pricing":{"prompt":"0.0000003","completion":"0.0000012"},"pricing_source":"openrouter","context_length":1048756,"max_completion_tokens":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"meituan","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sambanova/Meta-Llama-3.3-70B-Instruct","name":"Meta-Llama-3.3-70B-Instruct","_provider":"sambanova","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"sambanova","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"microsoft/phi-4","name":"Microsoft: Phi 4","description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","_provider":"microsoft","pricing":{"prompt":"0.00000007","completion":"0.00000014"},"pricing_source":"openrouter","context_length":16384,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"microsoft","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m1","name":"MiniMax: MiniMax M1","description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","_provider":"minimax","pricing":{"prompt":"0.00000055","completion":"0.0000022"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":40000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m2","name":"MiniMax: MiniMax M2","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","_provider":"minimax","pricing":{"prompt":"0.000000255","completion":"0.00000102"},"pricing_source":"openrouter","context_length":204800,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m2-her","name":"MiniMax: MiniMax M2-her","description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","_provider":"minimax","pricing":{"prompt":"0.0000003","completion":"0.0000012"},"pricing_source":"openrouter","context_length":65536,"max_completion_tokens":2048,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","temperature","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m2.1","name":"MiniMax: MiniMax M2.1","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","_provider":"minimax","pricing":{"prompt":"0.0000003","completion":"0.0000012"},"pricing_source":"openrouter","context_length":204800,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m3","name":"MiniMax: MiniMax M3","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","_provider":"minimax","pricing":{"prompt":"0.0000003","completion":"0.0000012"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":512000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-m3:batch","name":"MiniMax: MiniMax M3 (batch)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","_provider":"minimax","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"pricing_source":"openrouter","context_length":524288,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"minimax/minimax-01","name":"MiniMax: MiniMax-01","description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","_provider":"minimax","pricing":{"prompt":"0.0000002","completion":"0.0000011"},"pricing_source":"openrouter","context_length":1000192,"max_completion_tokens":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","temperature","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"minimax","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"moonshot-v1-128k","name":"moonshot-v1-128k","_provider":"moonshot","pricing":{"prompt":"0.0000002","completion":"0.0000005"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-128k-vision-preview","name":"moonshot-v1-128k-vision-preview","_provider":"moonshot","pricing":{"prompt":"0.0000002","completion":"0.0000005"},"context_length":262144,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-32k","name":"moonshot-v1-32k","_provider":"moonshot","pricing":{"prompt":"0.0000001","completion":"0.0000003"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-32k-vision-preview","name":"moonshot-v1-32k-vision-preview","_provider":"moonshot","pricing":{"prompt":"0.0000001","completion":"0.0000003"},"context_length":262144,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-8k","name":"moonshot-v1-8k","_provider":"moonshot","pricing":{"prompt":"0.00000002","completion":"0.0000002"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-8k-vision-preview","name":"moonshot-v1-8k-vision-preview","_provider":"moonshot","pricing":{"prompt":"0.00000002","completion":"0.0000002"},"context_length":262144,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"pricing_source":"portkey_models","routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshot-v1-auto","name":"moonshot-v1-auto","_provider":"moonshot","pricing":{"prompt":"0","completion":"0"},"context_length":262144,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"moonshot","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"moonshotai/kimi-k2","name":"MoonshotAI: Kimi K2 0711","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","_provider":"moonshotai","pricing":{"prompt":"0.00000057","completion":"0.0000023"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":100352,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshotai/kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","_provider":"moonshotai","pricing":{"prompt":"0.0000006","completion":"0.0000025"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":100352,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshotai/kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","_provider":"moonshotai","pricing":{"prompt":"0.0000006","completion":"0.0000025"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":100352,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"moonshotai/kimi-k2.7-code:batch","name":"MoonshotAI: Kimi K2.7 Code (batch)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","_provider":"moonshotai","pricing":{"prompt":"0.000000475","completion":"0.000002"},"pricing_source":"openrouter","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"moonshot","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"morph/morph-v3-fast","name":"Morph: Morph V3 Fast","description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","_provider":"morph","pricing":{"prompt":"0.0000008","completion":"0.0000012"},"pricing_source":"openrouter","context_length":81920,"max_completion_tokens":38000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature"],"routeability":{"status":"unrouteable","routeable":false,"provider":"morph","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"morph/morph-v3-large","name":"Morph: Morph V3 Large","description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","_provider":"morph","pricing":{"prompt":"0.0000009","completion":"0.0000019"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["logprobs","max_tokens","response_format","stop","structured_outputs","temperature","top_logprobs"],"routeability":{"status":"unrouteable","routeable":false,"provider":"morph","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","_provider":"gryphe","pricing":{"prompt":"0.00000008","completion":"0.00000011"},"pricing_source":"openrouter","context_length":8192,"max_completion_tokens":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"gryphe","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nex-agi/nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","_provider":"nex-agi","pricing":{"prompt":"0.000000025","completion":"0.0000001"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nex-agi","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nex-agi/nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","_provider":"nex-agi","pricing":{"prompt":"0.00000025","completion":"0.000001"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nex-agi","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","_provider":"nousresearch","pricing":{"prompt":"0.000001","completion":"0.000001"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nousresearch","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","_provider":"nousresearch","pricing":{"prompt":"0.0000007","completion":"0.0000007"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nousresearch","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nousresearch/hermes-4-405b","name":"Nous: Hermes 4 405B","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","_provider":"nousresearch","pricing":{"prompt":"0.000001","completion":"0.000003"},"pricing_source":"openrouter","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nousresearch","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nousresearch/hermes-4-70b","name":"Nous: Hermes 4 70B","description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","_provider":"nousresearch","pricing":{"prompt":"0.00000013","completion":"0.0000004"},"pricing_source":"openrouter","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"nousresearch","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"nvidia/nv-mistralai/mistral-nemo-12b-instruct","name":"nv-mistralai/mistral-nemo-12b-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-3-nano-30b-a3b:free","name":"NVIDIA: Nemotron 3 Nano 30B A3B (free)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"NVIDIA: Nemotron 3 Nano Omni (free)","description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":65536,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3-ultra-550b-a55b:batch","name":"NVIDIA: Nemotron 3 Ultra (batch)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","_provider":"nvidia","pricing":{"prompt":"0.0000003","completion":"0.0000018"},"pricing_source":"openrouter","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"NVIDIA: Nemotron 3 Ultra (free)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3.5-content-safety:free","name":"NVIDIA: Nemotron 3.5 Content Safety (free)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":8192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-nano-12b-v2-vl:free","name":"NVIDIA: Nemotron Nano 12B 2 VL (free)","description":"NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":128000,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-nano-9b-v2:free","name":"NVIDIA: Nemotron Nano 9B V2 (free)","description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/llama-3.1-nemoguard-8b-content-safety","name":"nvidia/llama-3.1-nemoguard-8b-content-safety","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemoguard-8b-topic-control","name":"nvidia/llama-3.1-nemoguard-8b-topic-control","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-51b-instruct","name":"nvidia/llama-3.1-nemotron-51b-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-70b-instruct","name":"nvidia/llama-3.1-nemotron-70b-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-nano-8b-v1","name":"nvidia/llama-3.1-nemotron-nano-8b-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-nano-vl-8b-v1","name":"nvidia/llama-3.1-nemotron-nano-vl-8b-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-safety-guard-8b-v3","name":"nvidia/llama-3.1-nemotron-safety-guard-8b-v3","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.1-nemotron-ultra-253b-v1","name":"nvidia/llama-3.1-nemotron-ultra-253b-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.2-nemoretriever-1b-vlm-embed-v1","name":"nvidia/llama-3.2-nemoretriever-1b-vlm-embed-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.2-nv-embedqa-1b-v1","name":"nvidia/llama-3.2-nv-embedqa-1b-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.3-nemotron-super-49b-v1","name":"nvidia/llama-3.3-nemotron-super-49b-v1","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","name":"nvidia/llama-3.3-nemotron-super-49b-v1.5","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-nemotron-embed-1b-v2","name":"nvidia/llama-nemotron-embed-1b-v2","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama-nemotron-embed-vl-1b-v2","name":"nvidia/llama-nemotron-embed-vl-1b-v2","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/llama3-chatqa-1.5-70b","name":"nvidia/llama3-chatqa-1.5-70b","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/mistral-nemo-minitron-8b-8k-instruct","name":"nvidia/mistral-nemo-minitron-8b-8k-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-3-embed-1b","name":"nvidia/nemotron-3-embed-1b","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"nvidia/nemotron-3-nano-30b-a3b","_provider":"nvidia","pricing":{"prompt":"0.00000005","completion":"0.0000002"},"context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"max_completion_tokens":262144,"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","name":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"nvidia/nemotron-3-ultra-550b-a55b","_provider":"nvidia","pricing":{"prompt":"0.0000006","completion":"0.0000036"},"context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing_source":"openrouter","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"nvidia","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"nvidia/nemotron-3.5-content-safety","name":"nvidia/nemotron-3.5-content-safety","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-4-340b-instruct","name":"nvidia/nemotron-4-340b-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-4-340b-reward","name":"nvidia/nemotron-4-340b-reward","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-mini-4b-instruct","name":"nvidia/nemotron-mini-4b-instruct","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"nvidia/nemotron-nano-12b-v2-vl","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"multimodal","input_modalities":["text","image"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-nano-3-30b-a3b","name":"nvidia/nemotron-nano-3-30b-a3b","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nemotron-parse","name":"nvidia/nemotron-parse","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nv-embedqa-mistral-7b-v2","name":"nvidia/nv-embedqa-mistral-7b-v2","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"embedding","input_modalities":["text"],"output_modalities":["embeddings"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"embeddings":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"nvidia/nvidia-nemotron-nano-9b-v2","name":"nvidia/nvidia-nemotron-nano-9b-v2","_provider":"nvidia","pricing":{"prompt":"0","completion":"0"},"context_length":32768,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"]},"routeability":{"status":"missing_pricing","routeable":false,"provider":"nvidia","mode":"system","reason":"provider key is configured but no real pricing source is available for this model","endpoints":{"chat_completions":{"status":"missing_pricing","routeable":false,"reason":"provider key is configured but no real pricing source is available for this model"}}},"routeable":false},{"id":"cerebras/gpt-oss-120b","name":"OpenAI GPT OSS","description":"This model excels at efficient reasoning across science, math, and coding applications. It's ideal for real-time coding assistance, processing large documents for Q&A and summarization, agentic research workflows, and regulated on-premises workloads.","_provider":"cerebras","pricing":{"prompt":"0.00000035","completion":"0.00000075"},"pricing_source":"cerebras_public","context_length":131072,"architecture":{"modality":"text","tokenizer":"GPT","instruct_type":"harmony","input_modalities":["text"],"output_modalities":["text"]},"supported_parameters":["frequency_penalty","logit_bias","max_completion_tokens","presence_penalty","seed","stop","temperature","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"cerebras","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"perceptron/perceptron-mk1","name":"Perceptron: Perceptron Mk1","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","_provider":"perceptron","pricing":{"prompt":"0.00000015","completion":"0.0000015"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":8192,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perceptron","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"perplexity/sonar","name":"Perplexity: Sonar","description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","_provider":"perplexity","pricing":{"prompt":"0.000001","completion":"0.000001"},"pricing_source":"openrouter","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perplexity","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"perplexity/sonar-deep-research","name":"Perplexity: Sonar Deep Research","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","_provider":"perplexity","pricing":{"prompt":"0.000002","completion":"0.000008"},"pricing_source":"openrouter","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perplexity","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"perplexity/sonar-pro","name":"Perplexity: Sonar Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","_provider":"perplexity","pricing":{"prompt":"0.000003","completion":"0.000015"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":8000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perplexity","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"perplexity/sonar-pro-search","name":"Perplexity: Sonar Pro Search","description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","_provider":"perplexity","pricing":{"prompt":"0.000003","completion":"0.000015"},"pricing_source":"openrouter","context_length":200000,"max_completion_tokens":8000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perplexity","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"perplexity/sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","_provider":"perplexity","pricing":{"prompt":"0.000002","completion":"0.000008"},"pricing_source":"openrouter","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"perplexity","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"poolside/laguna-s-2.1","name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","_provider":"poolside","pricing":{"prompt":"0.00000009","completion":"0.00000018"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"routeability":{"status":"unrouteable","routeable":false,"provider":"poolside","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"poolside/laguna-s-2.1:free","name":"Poolside: Laguna S 2.1 (free)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","_provider":"poolside","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"routeability":{"status":"unrouteable","routeable":false,"provider":"poolside","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"poolside/laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","_provider":"poolside","pricing":{"prompt":"0.00000006","completion":"0.00000012"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"routeability":{"status":"unrouteable","routeable":false,"provider":"poolside","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"poolside/laguna-xs-2.1:free","name":"Poolside: Laguna XS 2.1 (free)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","_provider":"poolside","pricing":{"prompt":"0","completion":"0"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"routeability":{"status":"unrouteable","routeable":false,"provider":"poolside","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","_provider":"rekaai","pricing":{"prompt":"0.0000001","completion":"0.0000001"},"pricing_source":"openrouter","context_length":16384,"max_completion_tokens":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"rekaai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","_provider":"rekaai","pricing":{"prompt":"0.0000001","completion":"0.0000002"},"pricing_source":"openrouter","context_length":65536,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"rekaai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"relace/relace-apply-3","name":"Relace: Relace Apply 3","description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","_provider":"relace","pricing":{"prompt":"0.00000085","completion":"0.00000125"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","seed","stop"],"routeability":{"status":"unrouteable","routeable":false,"provider":"relace","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"relace/relace-search","name":"Relace: Relace Search","description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","_provider":"relace","pricing":{"prompt":"0.000001","completion":"0.000003"},"pricing_source":"openrouter","context_length":256000,"max_completion_tokens":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"relace","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","_provider":"undi95","pricing":{"prompt":"0.00000045","completion":"0.00000065"},"pricing_source":"openrouter","context_length":6144,"max_completion_tokens":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"undi95","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sakana/fugu-ultra","name":"Sakana: Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","_provider":"sakana","pricing":{"prompt":"0.000005","completion":"0.00003"},"pricing_source":"openrouter","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"routeability":{"status":"unrouteable","routeable":false,"provider":"sakana","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sao10k/l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","_provider":"sao10k","pricing":{"prompt":"0.00000004","completion":"0.00000005"},"pricing_source":"openrouter","context_length":8192,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"sao10k","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sao10k/l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","_provider":"sao10k","pricing":{"prompt":"0.00000085","completion":"0.00000085"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"sao10k","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"sao10k/l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","_provider":"sao10k","pricing":{"prompt":"0.00000065","completion":"0.00000075"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"sao10k","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"x-ai/grok-4.3","name":"SpaceXAI: Grok 4.3","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","_provider":"x-ai","pricing":{"prompt":"0.00000125","completion":"0.0000025"},"pricing_source":"openrouter","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"x-ai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"x-ai/grok-4.5","name":"SpaceXAI: Grok 4.5","description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","_provider":"x-ai","pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_source":"openrouter","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"x-ai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"x-ai/grok-build-0.1","name":"SpaceXAI: Grok Build 0.1","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","_provider":"x-ai","pricing":{"prompt":"0.000001","completion":"0.000002"},"pricing_source":"openrouter","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"x-ai","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"stepfun/step-3.5-flash","name":"StepFun: Step 3.5 Flash","description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","_provider":"stepfun","pricing":{"prompt":"0.0000001","completion":"0.0000003"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"stepfun","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"stepfun/step-3.7-flash","name":"StepFun: Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","_provider":"stepfun","pricing":{"prompt":"0.0000002","completion":"0.00000115"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":256000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"stepfun","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"tencent/hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","_provider":"tencent","pricing":{"prompt":"0.00000014","completion":"0.00000057"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"tencent","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"tencent/hy3","name":"Tencent: Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","_provider":"tencent","pricing":{"prompt":"0.000000132","completion":"0.000000528"},"pricing_source":"openrouter","context_length":262144,"max_completion_tokens":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"tencent","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"tencent/hy3-preview","name":"Tencent: Hy3 preview","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","_provider":"tencent","pricing":{"prompt":"0.000000063","completion":"0.00000021"},"pricing_source":"openrouter","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"tencent","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","_provider":"thedrummer","pricing":{"prompt":"0.0000003","completion":"0.0000005"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thedrummer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thedrummer/rocinante-12b","name":"TheDrummer: Rocinante 12B","description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","_provider":"thedrummer","pricing":{"prompt":"0.00000025","completion":"0.0000005"},"pricing_source":"openrouter","context_length":65536,"max_completion_tokens":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thedrummer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thedrummer/skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","_provider":"thedrummer","pricing":{"prompt":"0.00000055","completion":"0.0000008"},"pricing_source":"openrouter","context_length":32768,"max_completion_tokens":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thedrummer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thedrummer/unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","_provider":"thedrummer","pricing":{"prompt":"0.0000004","completion":"0.0000004"},"pricing_source":"openrouter","context_length":1024000,"max_completion_tokens":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thedrummer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thinkingmachines/inkling","name":"Thinking Machines: Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","_provider":"thinkingmachines","pricing":{"prompt":"0.00000095","completion":"0.00000405"},"pricing_source":"openrouter","context_length":1048576,"max_completion_tokens":262144,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thinkingmachines","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thinkingmachines/inkling:batch","name":"Thinking Machines: Inkling (batch)","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","_provider":"thinkingmachines","pricing":{"prompt":"0.0000005","completion":"0.000002025"},"pricing_source":"openrouter","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thinkingmachines","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"thinkingmachines/inkling-small","name":"Thinking Machines: Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","_provider":"thinkingmachines","pricing":{"prompt":"0.00000045","completion":"0.0000012"},"pricing_source":"openrouter","context_length":524288,"max_completion_tokens":262144,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"thinkingmachines","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"upstage/solar-pro-3","name":"Upstage: Solar Pro 3","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","_provider":"upstage","pricing":{"prompt":"0.00000015","completion":"0.0000006"},"pricing_source":"openrouter","context_length":131072,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"upstage","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","_provider":"cognitivecomputations","pricing":{"prompt":"0.0000002","completion":"0.0000009"},"pricing_source":"openrouter","context_length":128000,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"cognitivecomputations","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","_provider":"microsoft","pricing":{"prompt":"0.00000062","completion":"0.00000062"},"pricing_source":"openrouter","context_length":65535,"max_completion_tokens":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"microsoft","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"writer/palmyra-x5","name":"Writer: Palmyra X5","description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","_provider":"writer","pricing":{"prompt":"0.0000006","completion":"0.000006"},"pricing_source":"openrouter","context_length":1040000,"max_completion_tokens":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"writer","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"xiaomi/mimo-v2.5","name":"Xiaomi: MiMo-V2.5","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","_provider":"xiaomi","pricing":{"prompt":"0.00000014","completion":"0.00000028"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":131072,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"xiaomi","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"xiaomi/mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","_provider":"xiaomi","pricing":{"prompt":"0.000000435","completion":"0.00000087"},"pricing_source":"openrouter","context_length":1050000,"max_completion_tokens":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"routeability":{"status":"unrouteable","routeable":false,"provider":"xiaomi","mode":"external","reason":"provider is present in the compatibility catalog but is not wired into chat_completions","endpoints":{"chat_completions":{"status":"unrouteable","routeable":false,"reason":"provider is present in the compatibility catalog but is not wired into chat_completions"}}},"routeable":false},{"id":"cerebras/zai-glm-4.7","name":"Z.ai GLM 4.7","description":"This model delivers strong coding performance with advanced reasoning capabilities, superior tool use, and enhanced real-world performance in agentic coding applications.","_provider":"cerebras","pricing":{"prompt":"0.00000225","completion":"0.00000275"},"pricing_source":"cerebras_public","context_length":131072,"architecture":{"modality":"text","tokenizer":"GLM","instruct_type":"glm","input_modalities":["text"],"output_modalities":["text"]},"supported_parameters":["frequency_penalty","logit_bias","max_completion_tokens","presence_penalty","seed","stop","temperature","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"cerebras","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"z-ai/glm-4.5v","name":"Z.ai: GLM 4.5V","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","_provider":"z-ai","pricing":{"prompt":"0.0000006","completion":"0.0000018"},"pricing_source":"openrouter","context_length":65536,"max_completion_tokens":16384,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true},{"id":"z-ai/glm-5.2:batch","name":"Z.ai: GLM 5.2 (batch)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","_provider":"z-ai","pricing":{"prompt":"0.0000007","completion":"0.0000022"},"pricing_source":"openrouter","context_length":512000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"routeability":{"status":"routeable","routeable":true,"provider":"zai","mode":"system","endpoints":{"chat_completions":{"status":"routeable","routeable":true}}},"routeable":true}]}