{"data":[{"id":"gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0022500000000000003,"internal_reasoning":0.0,"input_cache_read":4.500000000000001e-09,"input_cache_write":5.625e-08,"max_prompt_cost":0.04725,"max_completion_cost":0.03456,"max_cost":0.07605},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":3.459718383536538,"internal_reasoning":0.0,"input_cache_read":6.919436767073076e-06,"input_cache_write":8.649295958841345e-05,"max_prompt_cost":72.6540860542673,"max_completion_cost":53.14127437112122,"max_cost":116.93848136353499},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro:batch","name":"OpenAI: GPT-5.6 Luna Pro (batch)","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.04725,"max_completion_cost":0.03456,"max_cost":0.07605},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073076e-06,"input_cache_write":0.0,"max_prompt_cost":72.6540860542673,"max_completion_cost":53.14127437112122,"max_cost":116.93848136353499},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna","name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0022500000000000003,"internal_reasoning":0.0,"input_cache_read":4.500000000000001e-09,"input_cache_write":5.625e-08,"max_prompt_cost":0.04725,"max_completion_cost":0.03456,"max_cost":0.07605},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":3.459718383536538,"internal_reasoning":0.0,"input_cache_read":6.919436767073076e-06,"input_cache_write":8.649295958841345e-05,"max_prompt_cost":72.6540860542673,"max_completion_cost":53.14127437112122,"max_cost":116.93848136353499},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna:batch","name":"OpenAI: GPT-5.6 Luna (batch)","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.04725,"max_completion_cost":0.03456,"max_cost":0.07605},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073076e-06,"input_cache_write":0.0,"max_prompt_cost":72.6540860542673,"max_completion_cost":53.14127437112122,"max_cost":116.93848136353499},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":2.7e-06,"request":0.0,"image":0.0,"web_search":0.0022500000000000003,"internal_reasoning":0.0,"input_cache_read":4.5e-08,"input_cache_write":5.625e-07,"max_prompt_cost":0.4725,"max_completion_cost":0.3456,"max_cost":0.7605},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.004151662060243845,"request":0.001,"image":0.0,"web_search":3.459718383536538,"internal_reasoning":0.0,"input_cache_read":6.919436767073075e-05,"input_cache_write":0.0008649295958841345,"max_prompt_cost":726.5408605426729,"max_completion_cost":531.4127437112122,"max_cost":1169.3848136353497},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro:batch","name":"OpenAI: GPT-5.6 Terra Pro (batch)","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":2.7e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.4725,"max_completion_cost":0.3456,"max_cost":0.7605},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.004151662060243845,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073075e-05,"input_cache_write":0.0,"max_prompt_cost":726.5408605426729,"max_completion_cost":531.4127437112122,"max_cost":1169.3848136353497},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra","name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":2.7e-06,"request":0.0,"image":0.0,"web_search":0.0022500000000000003,"internal_reasoning":0.0,"input_cache_read":4.5e-08,"input_cache_write":5.625e-07,"max_prompt_cost":0.4725,"max_completion_cost":0.3456,"max_cost":0.7605},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.004151662060243845,"request":0.001,"image":0.0,"web_search":3.459718383536538,"internal_reasoning":0.0,"input_cache_read":6.919436767073075e-05,"input_cache_write":0.0008649295958841345,"max_prompt_cost":726.5408605426729,"max_completion_cost":531.4127437112122,"max_cost":1169.3848136353497},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra:batch","name":"OpenAI: GPT-5.6 Terra (batch)","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":2.7e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.4725,"max_completion_cost":0.3456,"max_cost":0.7605},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.004151662060243845,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073075e-05,"input_cache_write":0.0,"max_prompt_cost":726.5408605426729,"max_completion_cost":531.4127437112122,"max_cost":1169.3848136353497},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":2.8125e-06,"max_prompt_cost":2.3625000000000003,"max_completion_cost":1.7280000000000002,"max_cost":3.8025},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.004324647979420672,"max_prompt_cost":3632.704302713365,"max_completion_cost":2657.0637185560613,"max_cost":5846.92406817675},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro:batch","name":"OpenAI: GPT-5.6 Sol Pro (batch)","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":1.1812500000000001,"max_completion_cost":0.8640000000000001,"max_cost":1.90125},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":1816.3521513566825,"max_completion_cost":1328.5318592780307,"max_cost":2923.462034088375},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol","name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":2.8125e-06,"max_prompt_cost":2.3625000000000003,"max_completion_cost":1.7280000000000002,"max_cost":3.8025},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.004324647979420672,"max_prompt_cost":3632.704302713365,"max_completion_cost":2657.0637185560613,"max_cost":5846.92406817675},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol:batch","name":"OpenAI: GPT-5.6 Sol (batch)","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":1.1812500000000001,"max_completion_cost":0.8640000000000001,"max_cost":1.90125},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":1816.3521513566825,"max_completion_cost":1328.5318592780307,"max_cost":2923.462034088375},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-chat-latest","name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":0.0,"max_prompt_cost":0.9,"max_completion_cost":1.7280000000000002,"max_cost":2.3400000000000003},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.0,"max_prompt_cost":1383.887353414615,"max_completion_cost":2657.0637185560613,"max_cost":3598.1071188779997},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-chat-latest-20260505","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3500000000000001e-05,"completion":8.1e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.35e-06,"input_cache_write":0.0,"max_prompt_cost":14.175,"max_completion_cost":10.368,"max_cost":22.815},"sats_pricing":{"prompt":0.02075831030121923,"completion":0.12454986180731537,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0020758310301219225,"input_cache_write":0.0,"max_prompt_cost":21796.225816280188,"max_completion_cost":15942.382311336367,"max_cost":35081.544409060494},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro:batch","name":"OpenAI: GPT-5.5 Pro (batch)","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.750000000000001e-06,"completion":4.05e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.0875,"max_completion_cost":5.184,"max_cost":11.4075},"sats_pricing":{"prompt":0.010379155150609614,"completion":0.06227493090365768,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10898.112908140094,"max_completion_cost":7971.1911556681835,"max_cost":17540.772204530247},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":0.0,"max_prompt_cost":2.3625000000000003,"max_completion_cost":1.7280000000000002,"max_cost":3.8025},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.0,"max_prompt_cost":3632.704302713365,"max_completion_cost":2657.0637185560613,"max_cost":5846.92406817675},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5:batch","name":"OpenAI: GPT-5.5 (batch)","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":1.1812500000000001,"max_completion_cost":0.8640000000000001,"max_cost":1.90125},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":1816.3521513566825,"max_completion_cost":1328.5318592780307,"max_cost":2923.462034088375},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.6e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":9e-07,"input_cache_write":0.0,"max_prompt_cost":0.9792,"max_completion_cost":0.8640000000000001,"max_cost":1.3824},"sats_pricing":{"prompt":0.005535549413658461,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0013838873534146152,"input_cache_write":0.0,"max_prompt_cost":1505.6694405151013,"max_completion_cost":1328.5318592780307,"max_cost":2125.650974844849},"per_request_limits":null,"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-image-2-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9e-08,"completion":5.625e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":9.000000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.036,"max_completion_cost":0.07200000000000001,"max_cost":0.09648000000000001},"sats_pricing":{"prompt":0.0001383887353414615,"completion":0.0008649295958841345,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":1.3838873534146153e-05,"input_cache_write":0.0,"max_prompt_cost":55.3554941365846,"max_completion_cost":110.71098827316922,"max_cost":148.35272428604677},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano:batch","name":"OpenAI: GPT-5.4 Nano (batch)","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":2.8125e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.018,"max_completion_cost":0.036000000000000004,"max_cost":0.048240000000000005},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.00043246479794206726,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073076e-06,"input_cache_write":0.0,"max_prompt_cost":27.6777470682923,"max_completion_cost":55.35549413658461,"max_cost":74.17636214302338},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.375e-07,"completion":2.025e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":3.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.135,"max_completion_cost":0.2592,"max_cost":0.351},"sats_pricing":{"prompt":0.0005189577575304806,"completion":0.003113746545182884,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":5.189577575304807e-05,"input_cache_write":0.0,"max_prompt_cost":207.58310301219228,"max_completion_cost":398.55955778340916,"max_cost":539.7160678316999},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini:batch","name":"OpenAI: GPT-5.4 Mini (batch)","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.6875e-07,"completion":1.0125e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.0675,"max_completion_cost":0.1296,"max_cost":0.1755},"sats_pricing":{"prompt":0.0002594788787652403,"completion":0.001556873272591442,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":2.5947887876524036e-05,"input_cache_write":0.0,"max_prompt_cost":103.79155150609614,"max_completion_cost":199.27977889170458,"max_cost":269.85803391584994},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3500000000000001e-05,"completion":8.1e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.35e-06,"input_cache_write":0.0,"max_prompt_cost":14.175,"max_completion_cost":10.368,"max_cost":22.815},"sats_pricing":{"prompt":0.02075831030121923,"completion":0.12454986180731537,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0020758310301219225,"input_cache_write":0.0,"max_prompt_cost":21796.225816280188,"max_completion_cost":15942.382311336367,"max_cost":35081.544409060494},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro:batch","name":"OpenAI: GPT-5.4 Pro (batch)","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.750000000000001e-06,"completion":4.05e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.0875,"max_completion_cost":5.184,"max_cost":11.4075},"sats_pricing":{"prompt":0.010379155150609614,"completion":0.06227493090365768,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10898.112908140094,"max_completion_cost":7971.1911556681835,"max_cost":17540.772204530247},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":1.1812500000000001,"max_completion_cost":0.8640000000000001,"max_cost":1.90125},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":1816.3521513566825,"max_completion_cost":1328.5318592780307,"max_cost":2923.462034088375},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4:batch","name":"OpenAI: GPT-5.4 (batch)","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":3.3750000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.5906250000000001,"max_completion_cost":0.43200000000000005,"max_cost":0.950625},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.005189577575304807,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841345e-05,"input_cache_write":0.0,"max_prompt_cost":908.1760756783412,"max_completion_cost":664.2659296390153,"max_cost":1461.7310170441874},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-chat","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-07,"completion":6.3e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.1008,"max_completion_cost":0.1032192,"max_cost":0.19111679999999998},"sats_pricing":{"prompt":0.0012109014342377882,"completion":0.009687211473902306,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00012109014342377883,"input_cache_write":0.0,"max_prompt_cost":154.9953835824369,"max_completion_cost":158.71527278841538,"max_cost":293.8712472723003},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.3-chat-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-codex","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-07,"completion":6.3e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.315,"max_completion_cost":0.8064,"max_cost":1.0206},"sats_pricing":{"prompt":0.0012109014342377882,"completion":0.009687211473902306,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00012109014342377883,"input_cache_write":0.0,"max_prompt_cost":484.3605736951153,"max_completion_cost":1239.963068659495,"max_cost":1569.3282587721735},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.3-codex-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14400000000000002,"max_completion_cost":0.073728,"max_cost":0.19929600000000003},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":221.42197654633844,"max_completion_cost":113.36805199172528,"max_cost":306.4480155401324},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-audio","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio-mini","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7e-07,"completion":1.08e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.03456,"max_completion_cost":0.01769472,"max_cost":0.047831040000000005},"sats_pricing":{"prompt":0.0004151662060243846,"completion":0.0016606648240975383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53.14127437112122,"max_completion_cost":27.208332478014068,"max_cost":73.54752372963178},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-audio-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-codex","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-07,"completion":6.3e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.315,"max_completion_cost":0.8064,"max_cost":1.0206},"sats_pricing":{"prompt":0.0012109014342377882,"completion":0.009687211473902306,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00012109014342377883,"input_cache_write":0.0,"max_prompt_cost":484.3605736951153,"max_completion_cost":1239.963068659495,"max_cost":1569.3282587721735},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-codex-20260114","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-07,"completion":6.3e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.1008,"max_completion_cost":0.1032192,"max_cost":0.19111679999999998},"sats_pricing":{"prompt":0.0012109014342377882,"completion":0.009687211473902306,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00012109014342377883,"input_cache_write":0.0,"max_prompt_cost":154.9953835824369,"max_completion_cost":158.71527278841538,"max_cost":293.8712472723003},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-chat-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.45e-06,"completion":7.56e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.78,"max_completion_cost":9.6768,"max_cost":12.2472},"sats_pricing":{"prompt":0.014530817210853458,"completion":0.11624653768682766,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5812.326884341383,"max_completion_cost":14879.556823913943,"max_cost":18831.939105266083},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro:batch","name":"OpenAI: GPT-5.2 Pro (batch)","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.725e-06,"completion":3.78e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.89,"max_completion_cost":4.8384,"max_cost":6.1236},"sats_pricing":{"prompt":0.007265408605426729,"completion":0.05812326884341383,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2906.1634421706917,"max_completion_cost":7439.7784119569715,"max_cost":9415.969552633042},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-07,"completion":6.3e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.315,"max_completion_cost":0.8064,"max_cost":1.0206},"sats_pricing":{"prompt":0.0012109014342377882,"completion":0.009687211473902306,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00012109014342377883,"input_cache_write":0.0,"max_prompt_cost":484.3605736951153,"max_completion_cost":1239.963068659495,"max_cost":1569.3282587721735},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2:batch","name":"OpenAI: GPT-5.2 (batch)","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.9375e-07,"completion":3.15e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":3.9375e-08,"input_cache_write":0.0,"max_prompt_cost":0.1575,"max_completion_cost":0.4032,"max_cost":0.5103},"sats_pricing":{"prompt":0.0006054507171188941,"completion":0.004843605736951153,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.0545071711889415e-05,"input_cache_write":0.0,"max_prompt_cost":242.18028684755765,"max_completion_cost":619.9815343297475,"max_cost":784.6641293860868},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.225,"max_completion_cost":0.5760000000000001,"max_cost":0.7290000000000001},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841345e-05,"input_cache_write":0.0,"max_prompt_cost":345.97183835365377,"max_completion_cost":885.6879061853538,"max_cost":1120.9487562658385},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.1-codex-max-20251204","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.225,"max_completion_cost":0.5760000000000001,"max_cost":0.7290000000000001},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841345e-05,"input_cache_write":0.0,"max_prompt_cost":345.97183835365377,"max_completion_cost":885.6879061853538,"max_cost":1120.9487562658385},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1:batch","name":"OpenAI: GPT-5.1 (batch)","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8125e-07,"completion":2.25e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.8125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1125,"max_completion_cost":0.28800000000000003,"max_cost":0.36450000000000005},"sats_pricing":{"prompt":0.00043246479794206726,"completion":0.003459718383536538,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":4.3246479794206724e-05,"input_cache_write":0.0,"max_prompt_cost":172.98591917682688,"max_completion_cost":442.8439530926769,"max_cost":560.4743781329192},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.85e-08,"input_cache_write":0.0,"max_prompt_cost":0.225,"max_completion_cost":0.5760000000000001,"max_cost":0.7290000000000001},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.995267797194998e-05,"input_cache_write":0.0,"max_prompt_cost":345.97183835365377,"max_completion_cost":885.6879061853538,"max_cost":1120.9487562658385},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.1-codex-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.3499999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.045,"max_completion_cost":0.1152,"max_cost":0.14579999999999999},"sats_pricing":{"prompt":0.0001729859191768269,"completion":0.0013838873534146152,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":2.0758310301219226e-05,"input_cache_write":0.0,"max_prompt_cost":69.19436767073076,"max_completion_cost":177.13758123707072,"max_cost":224.18975125316763},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.375e-08,"completion":1.35e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.00442368,"max_completion_cost":0.00884736,"max_cost":0.0110592},"sats_pricing":{"prompt":5.189577575304807e-05,"completion":0.0002075831030121923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5947887876524036e-05,"input_cache_write":0.0,"max_prompt_cost":6.802083119503517,"max_completion_cost":13.604166239007034,"max_cost":17.00520779875879},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-oss-safeguard-20b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":0.45,"max_completion_cost":0.1152,"max_cost":0.4212},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.0013838873534146152,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":691.9436767073075,"max_completion_cost":177.13758123707072,"max_cost":647.6592813980399},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-image-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-06,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-07,"input_cache_write":0.0,"max_prompt_cost":1.8,"max_completion_cost":0.5760000000000001,"max_cost":1.8},"sats_pricing":{"prompt":0.006919436767073076,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0008649295958841345,"input_cache_write":0.0,"max_prompt_cost":2767.77470682923,"max_completion_cost":885.6879061853538,"max_cost":2767.77470682923},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-image","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.750000000000001e-06,"completion":5.4000000000000005e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.7,"max_completion_cost":6.912000000000001,"max_cost":8.748000000000001},"sats_pricing":{"prompt":0.010379155150609614,"completion":0.08303324120487691,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4151.662060243846,"max_completion_cost":10628.254874224245,"max_cost":13451.38507519006},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro:batch","name":"OpenAI: GPT-5 Pro (batch)","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.3750000000000003e-06,"completion":2.7000000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.35,"max_completion_cost":3.4560000000000004,"max_cost":4.3740000000000006},"sats_pricing":{"prompt":0.005189577575304807,"completion":0.04151662060243846,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2075.831030121923,"max_completion_cost":5314.127437112123,"max_cost":6725.69253759503},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-codex:batch","name":"OpenAI: GPT-5 Codex (batch)","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8125e-07,"completion":2.25e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.8125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1125,"max_completion_cost":0.28800000000000003,"max_cost":0.36450000000000005},"sats_pricing":{"prompt":0.00043246479794206726,"completion":0.003459718383536538,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":4.3246479794206724e-05,"input_cache_write":0.0,"max_prompt_cost":172.98591917682688,"max_completion_cost":442.8439530926769,"max_cost":560.4743781329192},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-codex","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.225,"max_completion_cost":0.5760000000000001,"max_cost":0.7290000000000001},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841345e-05,"input_cache_write":0.0,"max_prompt_cost":345.97183835365377,"max_completion_cost":885.6879061853538,"max_cost":1120.9487562658385},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5:batch","name":"OpenAI: GPT-5 (batch)","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8125e-07,"completion":2.25e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.8125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1125,"max_completion_cost":0.28800000000000003,"max_cost":0.36450000000000005},"sats_pricing":{"prompt":0.00043246479794206726,"completion":0.003459718383536538,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":4.3246479794206724e-05,"input_cache_write":0.0,"max_prompt_cost":172.98591917682688,"max_completion_cost":442.8439530926769,"max_cost":560.4743781329192},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.045,"max_completion_cost":0.1152,"max_cost":0.14579999999999999},"sats_pricing":{"prompt":0.0001729859191768269,"completion":0.0013838873534146152,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":1.7298591917682687e-05,"input_cache_write":0.0,"max_prompt_cost":69.19436767073076,"max_completion_cost":177.13758123707072,"max_cost":224.18975125316763},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini:batch","name":"OpenAI: GPT-5 Mini (batch)","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-08,"completion":4.5e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-09,"input_cache_write":0.0,"max_prompt_cost":0.0225,"max_completion_cost":0.0576,"max_cost":0.07289999999999999},"sats_pricing":{"prompt":8.649295958841345e-05,"completion":0.0006919436767073076,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841344e-06,"input_cache_write":0.0,"max_prompt_cost":34.59718383536538,"max_completion_cost":88.56879061853536,"max_cost":112.09487562658381},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-08,"completion":1.8e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.2500000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.009,"max_completion_cost":0.023039999999999998,"max_cost":0.02916},"sats_pricing":{"prompt":3.4597183835365375e-05,"completion":0.000276777470682923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":3.459718383536538e-06,"input_cache_write":0.0,"max_prompt_cost":13.83887353414615,"max_completion_cost":35.42751624741415,"max_cost":44.837950250633526},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano:batch","name":"OpenAI: GPT-5 Nano (batch)","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-08,"completion":9e-08,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1250000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.0045,"max_completion_cost":0.011519999999999999,"max_cost":0.01458},"sats_pricing":{"prompt":1.7298591917682687e-05,"completion":0.0001383887353414615,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":1.729859191768269e-06,"input_cache_write":0.0,"max_prompt_cost":6.919436767073075,"max_completion_cost":17.713758123707073,"max_cost":22.418975125316763},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.665e-08,"completion":7.649999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0021823488,"max_completion_cost":0.010027007999999999,"max_cost":0.010027007999999999},"sats_pricing":{"prompt":2.560191603817038e-05,"completion":0.00011763042504024227,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.355694338955068,"max_completion_cost":15.418055070874635,"max_cost":15.418055070874635},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-oss-120b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3499999999999998e-08,"completion":5.85e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3499999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.0017694719999999998,"max_completion_cost":0.007667712,"max_cost":0.007667712},"sats_pricing":{"prompt":2.0758310301219226e-05,"completion":8.995267797194998e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0758310301219226e-05,"input_cache_write":0.0,"max_prompt_cost":2.7208332478014063,"max_completion_cost":11.790277407139428,"max_cost":11.790277407139428},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-oss-20b","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9e-06,"completion":3.6e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.8,"max_completion_cost":3.6,"max_cost":4.5},"sats_pricing":{"prompt":0.013838873534146152,"completion":0.05535549413658461,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2767.77470682923,"max_completion_cost":5535.54941365846,"max_cost":6919.436767073075},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro:batch","name":"OpenAI: o3 Pro (batch)","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-06,"completion":1.8e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.9,"max_completion_cost":1.8,"max_cost":2.25},"sats_pricing":{"prompt":0.006919436767073076,"completion":0.027677747068292305,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1383.887353414615,"max_completion_cost":2767.77470682923,"max_cost":3459.7183835365377},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.95e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.2375e-07,"input_cache_write":0.0,"max_prompt_cost":0.099,"max_completion_cost":0.198,"max_cost":0.2475},"sats_pricing":{"prompt":0.0007611380443780383,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00019028451109450958,"input_cache_write":0.0,"max_prompt_cost":152.22760887560767,"max_completion_cost":304.45521775121534,"max_cost":380.56902218901917},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high:batch","name":"OpenAI: o4 Mini High (batch)","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.475e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":6.1875e-08,"input_cache_write":0.0,"max_prompt_cost":0.0495,"max_completion_cost":0.099,"max_cost":0.12375},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":9.514225554725479e-05,"input_cache_write":0.0,"max_prompt_cost":76.11380443780384,"max_completion_cost":152.22760887560767,"max_cost":190.28451109450958},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9e-07,"completion":3.6e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":0.0,"max_prompt_cost":0.18,"max_completion_cost":0.36,"max_cost":0.44999999999999996},"sats_pricing":{"prompt":0.0013838873534146152,"completion":0.005535549413658461,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.0,"max_prompt_cost":276.777470682923,"max_completion_cost":553.554941365846,"max_cost":691.9436767073075},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3:batch","name":"OpenAI: o3 (batch)","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":1.8e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":0.09,"max_completion_cost":0.18,"max_cost":0.22499999999999998},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.0027677747068292303,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":138.3887353414615,"max_completion_cost":276.777470682923,"max_cost":345.97183835365377},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.95e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.2375e-07,"input_cache_write":0.0,"max_prompt_cost":0.099,"max_completion_cost":0.198,"max_cost":0.2475},"sats_pricing":{"prompt":0.0007611380443780383,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00019028451109450958,"input_cache_write":0.0,"max_prompt_cost":152.22760887560767,"max_completion_cost":304.45521775121534,"max_cost":380.56902218901917},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini:batch","name":"OpenAI: o4 Mini (batch)","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.475e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":6.1875e-08,"input_cache_write":0.0,"max_prompt_cost":0.0495,"max_completion_cost":0.099,"max_cost":0.12375},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":9.514225554725479e-05,"input_cache_write":0.0,"max_prompt_cost":76.11380443780384,"max_completion_cost":152.22760887560767,"max_cost":190.28451109450958},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9e-07,"completion":3.6e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-07,"input_cache_write":0.0,"max_prompt_cost":0.9428184,"max_completion_cost":0.1179648,"max_cost":1.031292},"sats_pricing":{"prompt":0.0013838873534146152,"completion":0.005535549413658461,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0003459718383536538,"input_cache_write":0.0,"max_prompt_cost":1449.7271781406687,"max_completion_cost":181.38888318676044,"max_cost":1585.7688405307395},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1:batch","name":"OpenAI: GPT-4.1 (batch)","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":1.8e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":0.0,"max_prompt_cost":0.4714092,"max_completion_cost":0.0589824,"max_cost":0.515646},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.0027677747068292303,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0001729859191768269,"input_cache_write":0.0,"max_prompt_cost":724.8635890703343,"max_completion_cost":90.69444159338022,"max_cost":792.8844202653697},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8e-07,"completion":7.2e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":4.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.18856367999999998,"max_completion_cost":0.02359296,"max_cost":0.20625839999999998},"sats_pricing":{"prompt":0.000276777470682923,"completion":0.001107109882731692,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":6.919436767073075e-05,"input_cache_write":0.0,"max_prompt_cost":289.94543562813374,"max_completion_cost":36.27777663735208,"max_cost":317.1537681061478},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini:batch","name":"OpenAI: GPT-4.1 Mini (batch)","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9e-08,"completion":3.6e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.09428183999999999,"max_completion_cost":0.01179648,"max_cost":0.10312919999999999},"sats_pricing":{"prompt":0.0001383887353414615,"completion":0.000553554941365846,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":3.4597183835365375e-05,"input_cache_write":0.0,"max_prompt_cost":144.97271781406687,"max_completion_cost":18.13888831867604,"max_cost":158.5768840530739},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":1.8e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.047140919999999996,"max_completion_cost":0.00589824,"max_cost":0.051564599999999995},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.000276777470682923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":1.7298591917682687e-05,"input_cache_write":0.0,"max_prompt_cost":72.48635890703343,"max_completion_cost":9.06944415933802,"max_cost":79.28844202653696},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano:batch","name":"OpenAI: GPT-4.1 Nano (batch)","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-08,"completion":9e-08,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":5.625e-09,"input_cache_write":0.0,"max_prompt_cost":0.023570459999999998,"max_completion_cost":0.00294912,"max_cost":0.025782299999999998},"sats_pricing":{"prompt":3.4597183835365375e-05,"completion":0.0001383887353414615,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":8.649295958841344e-06,"input_cache_write":0.0,"max_prompt_cost":36.24317945351672,"max_completion_cost":4.53472207966901,"max_cost":39.64422101326848},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.75e-05,"completion":0.00027,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.5,"max_completion_cost":27.0,"max_cost":33.75},"sats_pricing":{"prompt":0.10379155150609613,"completion":0.41516620602438453,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":20758.310301219226,"max_completion_cost":41516.62060243845,"max_cost":51895.77575304807},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro:batch","name":"OpenAI: o1-pro (batch)","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.375e-05,"completion":0.000135,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.75,"max_completion_cost":13.5,"max_cost":16.875},"sats_pricing":{"prompt":0.051895775753048067,"completion":0.20758310301219227,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10379.155150609613,"max_completion_cost":20758.310301219226,"max_cost":25947.887876524033},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.95e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.475e-07,"input_cache_write":0.0,"max_prompt_cost":0.099,"max_completion_cost":0.198,"max_cost":0.2475},"sats_pricing":{"prompt":0.0007611380443780383,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00038056902218901916,"input_cache_write":0.0,"max_prompt_cost":152.22760887560767,"max_completion_cost":304.45521775121534,"max_cost":380.56902218901917},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high:batch","name":"OpenAI: o3 Mini High (batch)","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.475e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.2375e-07,"input_cache_write":0.0,"max_prompt_cost":0.0495,"max_completion_cost":0.099,"max_cost":0.12375},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00019028451109450958,"input_cache_write":0.0,"max_prompt_cost":76.11380443780384,"max_completion_cost":152.22760887560767,"max_cost":190.28451109450958},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.95e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.475e-07,"input_cache_write":0.0,"max_prompt_cost":0.099,"max_completion_cost":0.198,"max_cost":0.2475},"sats_pricing":{"prompt":0.0007611380443780383,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00038056902218901916,"input_cache_write":0.0,"max_prompt_cost":152.22760887560767,"max_completion_cost":304.45521775121534,"max_cost":380.56902218901917},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini:batch","name":"OpenAI: o3 Mini (batch)","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.475e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.2375e-07,"input_cache_write":0.0,"max_prompt_cost":0.0495,"max_completion_cost":0.099,"max_cost":0.12375},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00019028451109450958,"input_cache_write":0.0,"max_prompt_cost":76.11380443780384,"max_completion_cost":152.22760887560767,"max_cost":190.28451109450958},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o1","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.750000000000001e-06,"completion":2.7000000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":3.3750000000000003e-06,"input_cache_write":0.0,"max_prompt_cost":1.35,"max_completion_cost":2.7,"max_cost":3.375},"sats_pricing":{"prompt":0.010379155150609614,"completion":0.04151662060243846,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.005189577575304807,"input_cache_write":0.0,"max_prompt_cost":2075.831030121923,"max_completion_cost":4151.662060243846,"max_cost":5189.5775753048065},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"o1:batch","name":"OpenAI: o1 (batch)","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.3750000000000003e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.6875000000000001e-06,"input_cache_write":0.0,"max_prompt_cost":0.675,"max_completion_cost":1.35,"max_cost":1.6875},"sats_pricing":{"prompt":0.005189577575304807,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0025947887876524036,"input_cache_write":0.0,"max_prompt_cost":1037.9155150609615,"max_completion_cost":2075.831030121923,"max_cost":2594.7887876524032},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.14400000000000002,"max_completion_cost":0.073728,"max_cost":0.19929600000000003},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0008649295958841345,"input_cache_write":0.0,"max_prompt_cost":221.42197654633844,"max_completion_cost":113.36805199172528,"max_cost":306.4480155401324},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-2024-11-20","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.14400000000000002,"max_completion_cost":0.073728,"max_cost":0.19929600000000003},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0008649295958841345,"input_cache_write":0.0,"max_prompt_cost":221.42197654633844,"max_completion_cost":113.36805199172528,"max_cost":306.4480155401324},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-2024-08-06","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini","name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.75e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.00864,"max_completion_cost":0.00442368,"max_cost":0.011957760000000001},"sats_pricing":{"prompt":0.00010379155150609615,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.189577575304807e-05,"input_cache_write":0.0,"max_prompt_cost":13.285318592780305,"max_completion_cost":6.802083119503517,"max_cost":18.386880932407944},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.75e-08,"completion":2.7e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.00864,"max_completion_cost":0.00442368,"max_cost":0.011957760000000001},"sats_pricing":{"prompt":0.00010379155150609615,"completion":0.0004151662060243846,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.189577575304807e-05,"input_cache_write":0.0,"max_prompt_cost":13.285318592780305,"max_completion_cost":6.802083119503517,"max_cost":18.386880932407944},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-mini-2024-07-18","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini:batch","name":"OpenAI: GPT-4o-mini (batch)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.375e-08,"completion":1.35e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":1.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.00432,"max_completion_cost":0.00221184,"max_cost":0.005978880000000001},"sats_pricing":{"prompt":5.189577575304807e-05,"completion":0.0002075831030121923,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":2.5947887876524036e-05,"input_cache_write":0.0,"max_prompt_cost":6.6426592963901525,"max_completion_cost":3.4010415597517585,"max_cost":9.193440466203972},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o","name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.14400000000000002,"max_completion_cost":0.073728,"max_cost":0.19929600000000003},"sats_pricing":{"prompt":0.001729859191768269,"completion":0.006919436767073076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0008649295958841345,"input_cache_write":0.0,"max_prompt_cost":221.42197654633844,"max_completion_cost":113.36805199172528,"max_cost":306.4480155401324},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.28800000000000003,"max_completion_cost":0.027648000000000002,"max_cost":0.30643200000000004},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":442.8439530926769,"max_completion_cost":42.51301949689698,"max_cost":471.18596609060825},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o-2024-05-13","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o:batch","name":"OpenAI: GPT-4o (batch)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":2.25e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":2.8125e-07,"input_cache_write":0.0,"max_prompt_cost":0.07200000000000001,"max_completion_cost":0.036864,"max_cost":0.09964800000000001},"sats_pricing":{"prompt":0.0008649295958841345,"completion":0.003459718383536538,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.00043246479794206726,"input_cache_write":0.0,"max_prompt_cost":110.71098827316922,"max_completion_cost":56.68402599586264,"max_cost":153.2240077700662},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5760000000000001,"max_completion_cost":0.055296000000000005,"max_cost":0.6128640000000001},"sats_pricing":{"prompt":0.006919436767073076,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":885.6879061853538,"max_completion_cost":85.02603899379396,"max_cost":942.3719321812165},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo:batch","name":"OpenAI: GPT-4 Turbo (batch)","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-06,"completion":6.750000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.28800000000000003,"max_completion_cost":0.027648000000000002,"max_cost":0.30643200000000004},"sats_pricing":{"prompt":0.003459718383536538,"completion":0.010379155150609614,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":442.8439530926769,"max_completion_cost":42.51301949689698,"max_cost":471.18596609060825},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-07,"completion":9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00184275,"max_completion_cost":0.0036855,"max_cost":0.0036855},"sats_pricing":{"prompt":0.0006919436767073076,"completion":0.0013838873534146152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8335093561164246,"max_completion_cost":5.667018712232849,"max_cost":5.667018712232849},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-3.5-turbo-0613","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5e-06,"completion":1.3500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5760000000000001,"max_completion_cost":0.055296000000000005,"max_cost":0.6128640000000001},"sats_pricing":{"prompt":0.006919436767073076,"completion":0.02075831030121923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":885.6879061853538,"max_completion_cost":85.02603899379396,"max_cost":942.3719321812165},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4-turbo-preview","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":6.75e-07,"completion":9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002764125,"max_completion_cost":0.0036855,"max_cost":0.0036855},"sats_pricing":{"prompt":0.0010379155150609613,"completion":0.0013838873534146152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.250264034174637,"max_completion_cost":5.667018712232849,"max_cost":5.667018712232849},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-3.5-turbo-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.35e-06,"completion":1.8e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02211975,"max_completion_cost":0.0073728,"max_cost":0.02396295},"sats_pricing":{"prompt":0.0020758310301219225,"completion":0.0027677747068292303,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.012491428547705,"max_completion_cost":11.336805199172527,"max_cost":36.846692728340834},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-3.5-turbo-16k","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo","name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.25e-07,"completion":6.75e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.003686625,"max_completion_cost":0.0027648,"max_cost":0.005529825},"sats_pricing":{"prompt":0.0003459718383536538,"completion":0.0010379155150609613,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.668748571424617,"max_completion_cost":4.251301949689697,"max_cost":8.50294987121775},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo:batch","name":"OpenAI: GPT-3.5 Turbo (batch)","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":3.375e-07,"request":0.0,"image":0.0,"web_search":0.0045000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0018433125,"max_completion_cost":0.0013824,"max_cost":0.0027649125},"sats_pricing":{"prompt":0.0001729859191768269,"completion":0.0005189577575304806,"request":0.001,"image":0.0,"web_search":6.919436767073076,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8343742857123084,"max_completion_cost":2.1256509748448487,"max_cost":4.251474935608875},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4","name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3500000000000001e-05,"completion":2.7000000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11057850000000001,"max_completion_cost":0.11059200000000001,"max_cost":0.1658745},"sats_pricing":{"prompt":0.02075831030121923,"completion":0.04151662060243846,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":170.03131967728672,"max_completion_cost":170.05207798758792,"max_cost":255.05735867108066},"per_request_limits":null,"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/gpt-4","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-ada-002","name":"OpenAI: Text Embedding Ada 002","created":1761865798,"description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.5e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00036864,"max_completion_cost":0.0,"max_cost":0.00036864},"sats_pricing":{"prompt":6.919436767073075e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5668402599586263,"max_completion_cost":0.0,"max_cost":0.5668402599586263},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/text-embedding-ada-002","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-large","name":"OpenAI: Text Embedding 3 Large","created":1761862866,"description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.85e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.000479232,"max_completion_cost":0.0,"max_cost":0.000479232},"sats_pricing":{"prompt":8.995267797194998e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.7368923379462142,"max_completion_cost":0.0,"max_cost":0.7368923379462142},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/text-embedding-3-large","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-small","name":"OpenAI: Text Embedding 3 Small","created":1761857455,"description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.000000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.372800000000001e-05,"max_completion_cost":0.0,"max_cost":7.372800000000001e-05},"sats_pricing":{"prompt":1.3838873534146153e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11336805199172528,"max_completion_cost":0.0,"max_cost":0.11336805199172528},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openai","canonical_slug":"openai/text-embedding-3-small","alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.2","name":"Meta: Muse Spark 1.2","created":1785959287,"description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":2.3375e-06,"request":0.0,"image":0.0,"web_search":0.0013750000000000001,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.7208960000000001,"max_completion_cost":2.4510464,"max_cost":2.4510464},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0035942629873407365,"request":0.001,"image":0.0,"web_search":2.114272345494551,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":1108.4876194746473,"max_completion_cost":3768.8579062138,"max_cost":3768.8579062138},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta/muse-spark-1.2-20260805","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.8-max","name":"Qwen: Qwen3.8 Max","created":1785731612,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":1.1,"max_completion_cost":0.4325376,"max_cost":1.3883584},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0021142723454945513,"max_prompt_cost":1691.4178763956409,"max_completion_cost":665.0925716847883,"max_cost":2134.8129241854995},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.8-max-20260803","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","created":1785606009,"description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":4.928e-08,"completion":9.856e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.856e-09,"input_cache_write":0.0,"max_prompt_cost":0.05167382528,"max_completion_cost":0.01291845632,"max_cost":0.058133053440000006},"sats_pricing":{"prompt":7.577552086252471e-05,"completion":0.00015155104172504943,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5155104172504942e-05,"input_cache_write":0.0,"max_prompt_cost":79.45639256394271,"max_completion_cost":19.86409814098568,"max_cost":89.38844163443555},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~deepseek/deepseek-v4-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":4.9500000000000006e-08,"completion":9.900000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.900000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.05190451200000001,"max_completion_cost":0.03801600000000001,"max_cost":0.07091251200000001},"sats_pricing":{"prompt":7.611380443780384e-05,"completion":0.00015222760887560768,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5222760887560768e-05,"input_cache_write":0.0,"max_prompt_cost":79.8111086021746,"max_completion_cost":58.455401808233354,"max_cost":109.03880950629129},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-flash-20260731","alias_ids":null,"forwarded_model_id":null},{"id":"inkling-small","name":"Thinking Machines: Inkling Small","created":1785443117,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.475e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.12976128,"max_completion_cost":0.17301504,"max_cost":0.23789568},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":199.52777150543648,"max_completion_cost":266.0370286739153,"max_cost":365.80091442663354},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-small-20260730","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-flash","name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.65e-08,"completion":7.150000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-09,"input_cache_write":2.0900000000000002e-08,"max_prompt_cost":0.016499999999999997,"max_completion_cost":0.004685824000000001,"max_cost":0.02010448},"sats_pricing":{"prompt":2.537126814593461e-05,"completion":0.00010994216196571666,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.0742536291869225e-06,"input_cache_write":3.213693965151718e-05,"max_prompt_cost":25.371268145934607,"max_completion_cost":7.205169526585207,"max_cost":30.913706243307846},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-flash-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5-fast","name":"Claude Opus 5 (Fast)","created":1784912546,"description":"Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.500000000000001e-06,"completion":2.7500000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.500000000000001,"max_completion_cost":3.5200000000000005,"max_cost":8.316},"sats_pricing":{"prompt":0.008457089381978205,"completion":0.04228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0008457089381978204,"input_cache_write":0.010571361727472757,"max_prompt_cost":8457.089381978205,"max_completion_cost":5412.537204466051,"max_cost":12787.119145551045},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-fast-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5","name":"Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5:batch","name":"Claude Opus 5 (batch)","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.7187500000000003e-06,"max_prompt_cost":1.3750000000000002,"max_completion_cost":0.8800000000000001,"max_cost":2.079},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0026428404318681892,"max_prompt_cost":2114.2723454945512,"max_completion_cost":1353.1343011165127,"max_cost":3196.7797863877613},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"ling-3.0-flash","name":"Ling-3.0-flash","created":1784818580,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1550000000000001e-08,"completion":3.465e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3100000000000005e-09,"input_cache_write":0.0,"max_prompt_cost":0.0030277632000000002,"max_completion_cost":0.0011354112,"max_cost":0.0037847040000000007},"sats_pricing":{"prompt":1.775988770215423e-05,"completion":5.3279663106462686e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.551977540430846e-06,"input_cache_write":0.0,"max_prompt_cost":4.6556480017935185,"max_completion_cost":1.7458680006725693,"max_cost":5.819560002241898},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-3.0-flash-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-s-2.1","name":"Poolside: Laguna S 2.1","created":1784652683,"description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.9500000000000006e-08,"completion":9.900000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.9500000000000005e-09,"input_cache_write":0.0,"max_prompt_cost":0.05190451200000001,"max_completion_cost":0.012976128000000002,"max_cost":0.05839257600000001},"sats_pricing":{"prompt":7.611380443780384e-05,"completion":0.00015222760887560768,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.611380443780384e-06,"input_cache_write":0.0,"max_prompt_cost":79.8111086021746,"max_completion_cost":19.95277715054365,"max_cost":89.78749717744643},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"poolside/laguna-s-2.1-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash","name":"Google: Gemini 3.6 Flash","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.125e-06,"request":0.0,"image":8.25e-07,"web_search":0.007700000000000001,"internal_reasoning":4.125e-06,"input_cache_read":8.25e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.8650752,"max_completion_cost":0.270336,"max_cost":1.081344},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.0063428170364836535,"request":0.001,"image":0.0012685634072967305,"web_search":11.839925134769487,"internal_reasoning":0.0063428170364836535,"input_cache_read":0.00012685634072967305,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":1330.1851433695765,"max_completion_cost":415.6828573029927,"max_cost":1662.7314292119709},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash:batch","name":"Google: Gemini 3.6 Flash (batch)","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":2.0625e-06,"request":0.0,"image":4.125e-07,"web_search":0.007700000000000001,"internal_reasoning":2.0625e-06,"input_cache_read":4.125e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.4325376,"max_completion_cost":0.135168,"max_cost":0.540672},"sats_pricing":{"prompt":0.0006342817036483653,"completion":0.0031714085182418267,"request":0.001,"image":0.0006342817036483653,"web_search":11.839925134769487,"internal_reasoning":0.0031714085182418267,"input_cache_read":6.342817036483653e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":665.0925716847883,"max_completion_cost":207.84142865149636,"max_cost":831.3657146059854},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite","name":"Google: Gemini 3.5 Flash Lite","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":1.65e-07,"web_search":0.007700000000000001,"internal_reasoning":1.3750000000000002e-06,"input_cache_read":1.65e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.17301504,"max_completion_cost":0.09011200000000001,"max_cost":0.2523136},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0021142723454945513,"request":0.001,"image":0.0002537126814593461,"web_search":11.839925134769487,"internal_reasoning":0.0021142723454945513,"input_cache_read":2.537126814593461e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":266.0370286739153,"max_completion_cost":138.56095243433091,"max_cost":387.97066681612654},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite:batch","name":"Google: Gemini 3.5 Flash Lite (batch)","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":6.875000000000001e-07,"request":0.0,"image":8.25e-08,"web_search":0.007700000000000001,"internal_reasoning":6.875000000000001e-07,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.045056000000000006,"max_cost":0.1261568},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0010571361727472757,"request":0.001,"image":0.00012685634072967305,"web_search":11.839925134769487,"internal_reasoning":0.0010571361727472757,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":69.28047621716546,"max_cost":193.98533340806327},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"longcat-2.0","name":"Meituan: LongCat 2.0","created":1784554658,"description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-09,"input_cache_write":0.0,"max_prompt_cost":0.17304474,"max_completion_cost":0.17301504,"max_cost":0.30280602},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.0742536291869225e-06,"input_cache_write":0.0,"max_prompt_cost":266.082696956578,"max_completion_cost":266.0370286739153,"max_cost":465.61046846201447},"per_request_limits":null,"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meituan/longcat-2.0-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"inkling","name":"Thinking Machines: Inkling","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.225000000000001e-07,"completion":2.2275000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.800000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.27394048000000004,"max_completion_cost":0.5839257600000001,"max_cost":0.7208960000000001},"sats_pricing":{"prompt":0.0008034234912879294,"completion":0.003425121199701173,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00013531343011165126,"input_cache_write":0.0,"max_prompt_cost":421.22529540036595,"max_completion_cost":897.8749717744643,"max_cost":1108.4876194746473},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"inkling:batch","name":"Thinking Machines: Inkling (batch)","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.1137500000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.675e-08,"input_cache_write":0.0,"max_prompt_cost":0.1441792,"max_completion_cost":0.5839257600000001,"max_cost":0.5839257600000001},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.0017125605998505864,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.188525974681472e-05,"input_cache_write":0.0,"max_prompt_cost":221.69752389492942,"max_completion_cost":897.8749717744643,"max_cost":897.8749717744643},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k3","name":"MoonshotAI: Kimi K3","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":1.7301504,"max_completion_cost":8.650752,"max_cost":8.650752},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0,"max_prompt_cost":2660.370286739153,"max_completion_cost":13301.851433695767,"max_cost":13301.851433695767},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k3-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.1","name":"Meta: Muse Spark 1.1","created":1784215741,"description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":2.3375e-06,"request":0.0,"image":0.0,"web_search":0.0013750000000000001,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.7208960000000001,"max_completion_cost":2.4510464,"max_cost":2.4510464},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0035942629873407365,"request":0.001,"image":0.0,"web_search":2.114272345494551,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":1108.4876194746473,"max_completion_cost":3768.8579062138,"max_cost":3768.8579062138},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta/muse-spark-1.1-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-air-v2.5","name":"Kwaipilot: KAT-Coder-Air V2.5","created":1783714590,"description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.02112,"max_completion_cost":0.0264,"max_cost":0.04092},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":32.4752232267963,"max_completion_cost":40.594029033495374,"max_cost":62.92074500191783},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-air-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","created":1783714589,"description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.0700000000000003e-07,"completion":1.6280000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.104192,"max_completion_cost":0.13024000000000002,"max_cost":0.20187200000000002},"sats_pricing":{"prompt":0.0006258246142663871,"completion":0.0025032984570655483,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":160.2111012521951,"max_completion_cost":200.2638765652439,"max_cost":310.409008676128},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-pro-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.5","name":"SpaceXAI: Grok 4.5","created":1783523154,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.55,"max_completion_cost":1.6500000000000001,"max_cost":1.6500000000000001},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0,"max_prompt_cost":845.7089381978204,"max_completion_cost":2537.1268145934614,"max_cost":2537.1268145934614},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.5-20260708","alias_ids":null,"forwarded_model_id":null},{"id":"grok-latest","name":"xAI: Grok Latest","created":1783519360,"description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.55,"max_completion_cost":1.6500000000000001,"max_cost":1.6500000000000001},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0,"max_prompt_cost":845.7089381978204,"max_completion_cost":2537.1268145934614,"max_cost":2537.1268145934614},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~x-ai/grok-latest","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","created":1783443096,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":7.7e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.900000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.05046272,"max_completion_cost":0.02523136,"max_cost":0.0630784},"sats_pricing":{"prompt":0.0005919962567384742,"completion":0.0011839925134769485,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015222760887560768,"input_cache_write":0.0,"max_prompt_cost":77.5941333632253,"max_completion_cost":38.79706668161265,"max_cost":96.99266670403163},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-3.0-mini-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0","name":"AionLabs: Aion-3.0","created":1783443095,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.125e-07,"input_cache_write":0.0,"max_prompt_cost":0.2162688,"max_completion_cost":0.1081344,"max_cost":0.270336},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0006342817036483653,"input_cache_write":0.0,"max_prompt_cost":332.5462858423941,"max_completion_cost":166.27314292119706,"max_cost":415.6828573029927},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-3.0-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"hy3","name":"Tencent: Hy3","created":1783344048,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.26e-08,"completion":2.904e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.815e-08,"input_cache_write":0.0,"max_prompt_cost":0.0190316544,"max_completion_cost":0.0371712,"max_cost":0.0469100544},"sats_pricing":{"prompt":0.00011163357984211229,"completion":0.00044653431936844917,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.7908394960528073e-05,"input_cache_write":0.0,"max_prompt_cost":29.264073154130685,"max_completion_cost":57.156392879161494,"max_cost":72.1313678135018},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hy3-20260706","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","created":1783002429,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-08,"completion":6.6e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.008650752,"max_completion_cost":0.002162688,"max_cost":0.009732095999999999},"sats_pricing":{"prompt":5.074253629186922e-05,"completion":0.00010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":13.301851433695765,"max_completion_cost":3.325462858423941,"max_cost":14.964582862907735},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"poolside/laguna-xs-2.1-20260625","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":5.500000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":1.1,"max_completion_cost":0.7040000000000001,"max_cost":1.6632000000000002},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.008457089381978205,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0021142723454945513,"max_prompt_cost":1691.4178763956409,"max_completion_cost":1082.50744089321,"max_cost":2557.423829110209},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5:batch","name":"Anthropic: Claude Sonnet 5 (batch)","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":2.7500000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":6.875000000000001e-07,"max_prompt_cost":0.55,"max_completion_cost":0.35200000000000004,"max_cost":0.8316000000000001},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.004228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0010571361727472757,"max_prompt_cost":845.7089381978204,"max_completion_cost":541.253720446605,"max_cost":1278.7119145551046},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","created":1782837225,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":8.25e-07,"request":0.0,"image":0.0,"web_search":0.007700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0090112,"max_completion_cost":0.0540672,"max_cost":0.0540672},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0012685634072967305,"request":0.001,"image":0.0,"web_search":11.839925134769487,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.856095243433089,"max_completion_cost":83.13657146059853,"max_cost":83.13657146059853},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","created":1782312964,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.375e-08,"completion":5.5e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3750000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.01441792,"max_cost":0.01441792},"sats_pricing":{"prompt":2.114272345494551e-05,"completion":8.457089381978204e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.114272345494551e-06,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":22.169752389492942,"max_cost":22.169752389492942},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nex-agi/nex-n2-mini","alias_ids":null,"forwarded_model_id":null},{"id":"fugu-ultra","name":"Sakana: Fugu Ultra","created":1782276303,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.65e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":0.0,"max_prompt_cost":2.7500000000000004,"max_completion_cost":2.112,"max_cost":4.51},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.025371268145934614,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0,"max_prompt_cost":4228.5446909891025,"max_completion_cost":3247.52232267963,"max_cost":6934.813293222127},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sakana/fugu-ultra-20260615","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","created":1781754065,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.007700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0360448,"max_completion_cost":0.0540672,"max_cost":0.0811008},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.002537126814593461,"request":0.001,"image":0.0,"web_search":11.839925134769487,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55.424380973732355,"max_completion_cost":83.13657146059853,"max_cost":124.7048571908978},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","created":1781754054,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":6.6e-06,"request":0.0,"image":1.1e-06,"web_search":0.007700000000000001,"internal_reasoning":6.6e-06,"input_cache_read":1.1e-07,"input_cache_write":2.0625e-07,"max_prompt_cost":0.0720896,"max_completion_cost":0.2162688,"max_cost":0.2523136},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.010148507258373844,"request":0.001,"image":0.0016914178763956407,"web_search":11.839925134769487,"internal_reasoning":0.010148507258373844,"input_cache_read":0.00016914178763956407,"input_cache_write":0.00031714085182418263,"max_prompt_cost":110.84876194746471,"max_completion_cost":332.5462858423941,"max_cost":387.97066681612654},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-pro-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1319000000000001e-07,"completion":3.5574000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1021000000000003e-08,"input_cache_write":0.0,"max_prompt_cost":0.11590656,"max_completion_cost":0.04553472000000001,"max_cost":0.14695296000000002},"sats_pricing":{"prompt":0.00017404689948111144,"completion":0.0005470045412263502,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.23229956179207e-05,"input_cache_write":0.0,"max_prompt_cost":178.2240250686581,"max_completion_cost":70.01658127697284,"max_cost":225.9626032120487},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2:batch","name":"Z.ai: GLM 5.2 (batch)","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":512000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":1.21e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.150000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.19712000000000002,"max_completion_cost":0.6195200000000001,"max_cost":0.6195200000000001},"sats_pricing":{"prompt":0.0005919962567384742,"completion":0.001860559664035205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010994216196571666,"input_cache_write":0.0,"max_prompt_cost":303.10208345009886,"max_completion_cost":952.6065479860249,"max_cost":952.6065479860249},"per_request_limits":null,"top_provider":{"context_length":512000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code","name":"MoonshotAI: Kimi K2.7 Code","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":1.925e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.10092544,"max_completion_cost":0.5046272,"max_cost":0.5046272},"sats_pricing":{"prompt":0.0005919962567384742,"completion":0.0029599812836923717,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":155.1882667264506,"max_completion_cost":775.9413336322531,"max_cost":775.9413336322531},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code:batch","name":"MoonshotAI: Kimi K2.7 Code (batch)","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.6125000000000004e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.225000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.06848512000000001,"max_completion_cost":0.2883584,"max_cost":0.2883584},"sats_pricing":{"prompt":0.0004017117456439647,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.034234912879294e-05,"input_cache_write":0.0,"max_prompt_cost":105.30632385009149,"max_completion_cost":443.39504778985884,"max_cost":443.39504778985884},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-latest","name":"Anthropic: Claude Fable Latest","created":1781029944,"description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.500000000000001e-06,"completion":2.7500000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.500000000000001,"max_completion_cost":3.5200000000000005,"max_cost":8.316},"sats_pricing":{"prompt":0.008457089381978205,"completion":0.04228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0008457089381978204,"input_cache_write":0.010571361727472757,"max_prompt_cost":8457.089381978205,"max_completion_cost":5412.537204466051,"max_cost":12787.119145551045},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-fable-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5","name":"Anthropic: Claude Fable 5","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.500000000000001e-06,"completion":2.7500000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.500000000000001,"max_completion_cost":3.5200000000000005,"max_cost":8.316},"sats_pricing":{"prompt":0.008457089381978205,"completion":0.04228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0008457089381978204,"input_cache_write":0.010571361727472757,"max_prompt_cost":8457.089381978205,"max_completion_cost":5412.537204466051,"max_cost":12787.119145551045},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5:batch","name":"Anthropic: Claude Fable 5 (batch)","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","created":1780937140,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.0360448,"max_completion_cost":0.1441792,"max_cost":0.1441792},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.114272345494551e-05,"input_cache_write":0.0,"max_prompt_cost":55.424380973732355,"max_completion_cost":221.69752389492942,"max_cost":221.69752389492942},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nex-agi/nex-n2-pro","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.16905504000000002,"max_completion_cost":1.01433024,"max_cost":1.01433024},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":259.947924318891,"max_completion_cost":1559.687545913346,"max_cost":1559.687545913346},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b:batch","name":"NVIDIA: Nemotron 3 Ultra (batch)","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.08452752000000001,"max_completion_cost":0.50716512,"max_cost":0.50716512},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":129.9739621594455,"max_completion_cost":779.843772956673,"max_cost":779.843772956673},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.7600000000000001e-07,"completion":7.040000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.5200000000000004e-08,"input_cache_write":2.2e-07,"max_prompt_cost":0.17600000000000002,"max_completion_cost":0.09227468800000001,"max_cost":0.24520601600000003},"sats_pricing":{"prompt":0.0002706268602233025,"completion":0.00108250744089321,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.412537204466051e-05,"input_cache_write":0.00033828357527912814,"max_prompt_cost":270.6268602233025,"max_completion_cost":141.88641529275483,"max_cost":377.0416716928687},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-plus-20260602","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.33792,"max_cost":0.33994752},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.074253629186922e-05,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":519.6035716287408,"max_cost":522.7211930585132},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3:batch","name":"MiniMax: MiniMax M3 (batch)","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":524288,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.04325376,"max_completion_cost":0.17301504,"max_cost":0.17301504},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":66.50925716847883,"max_completion_cost":266.0370286739153,"max_cost":266.0370286739153},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.7-flash","name":"StepFun: Step 3.7 Flash","created":1779985069,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":6.325000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.02816,"max_completion_cost":0.16192,"max_cost":0.16192},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0009725652789274935,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3828357527912815e-05,"input_cache_write":0.0,"max_prompt_cost":43.300297635728406,"max_completion_cost":248.97671140543832,"max_cost":248.97671140543832},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":256000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"stepfun/step-3.7-flash-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","created":1779913703,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.500000000000001e-06,"completion":2.7500000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.500000000000001,"max_completion_cost":3.5200000000000005,"max_cost":8.316},"sats_pricing":{"prompt":0.008457089381978205,"completion":0.04228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0008457089381978204,"input_cache_write":0.010571361727472757,"max_prompt_cost":8457.089381978205,"max_completion_cost":5412.537204466051,"max_cost":12787.119145551045},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-fast-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8","name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8:batch","name":"Anthropic: Claude Opus 4.8 (batch)","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.7187500000000003e-06,"max_prompt_cost":1.3750000000000002,"max_completion_cost":0.8800000000000001,"max_cost":2.079},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0026428404318681892,"max_prompt_cost":2114.2723454945512,"max_completion_cost":1353.1343011165127,"max_cost":3196.7797863877613},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-max","name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":8.112500000000001e-07,"completion":2.43375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6225000000000001e-07,"input_cache_write":1.0140625e-06,"max_prompt_cost":0.81125,"max_completion_cost":0.31899648,"max_cost":1.02391432},"sats_pricing":{"prompt":0.0012474206838417852,"completion":0.0037422620515253553,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000249484136768357,"input_cache_write":0.0015592758548022313,"max_prompt_cost":1247.4206838417851,"max_completion_cost":490.50577161753137,"max_cost":1574.424531586806},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-max-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"grok-build-0.1","name":"SpaceXAI: Grok Build 0.1","created":1779298123,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.1408,"max_completion_cost":0.2816,"max_cost":0.2816},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":216.50148817864203,"max_completion_cost":433.00297635728407,"max_cost":433.00297635728407},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-build-0.1-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash","name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.950000000000001e-06,"request":0.0,"image":8.25e-07,"web_search":0.007700000000000001,"internal_reasoning":4.950000000000001e-06,"input_cache_read":8.25e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.8650752,"max_completion_cost":0.32440320000000006,"max_cost":1.1354112},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.007611380443780385,"request":0.001,"image":0.0012685634072967305,"web_search":11.839925134769487,"internal_reasoning":0.007611380443780385,"input_cache_read":0.00012685634072967305,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":1330.1851433695765,"max_completion_cost":498.8194287635913,"max_cost":1745.8680006725692},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash:batch","name":"Google: Gemini 3.5 Flash (batch)","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":2.4750000000000004e-06,"request":0.0,"image":4.125e-07,"web_search":0.007700000000000001,"internal_reasoning":2.4750000000000004e-06,"input_cache_read":4.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.4325376,"max_completion_cost":0.16220160000000003,"max_cost":0.5677056},"sats_pricing":{"prompt":0.0006342817036483653,"completion":0.0038056902218901924,"request":0.001,"image":0.0006342817036483653,"web_search":11.839925134769487,"internal_reasoning":0.0038056902218901924,"input_cache_read":6.342817036483653e-05,"input_cache_write":0.0,"max_prompt_cost":665.0925716847883,"max_completion_cost":249.40971438179565,"max_cost":872.9340003362846},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.65e-05,"completion":8.25e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.65e-06,"input_cache_write":2.0625e-05,"max_prompt_cost":16.5,"max_completion_cost":10.56,"max_cost":24.948},"sats_pricing":{"prompt":0.025371268145934614,"completion":0.12685634072967306,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.002537126814593461,"input_cache_write":0.031714085182418264,"max_prompt_cost":25371.26814593461,"max_completion_cost":16237.611613398152,"max_cost":38361.35743665313},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"perceptron-mk1","name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":8.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00270336,"max_completion_cost":0.0067584,"max_cost":0.008785920000000001},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.156828573029927,"max_completion_cost":10.392071432574816,"max_cost":13.509692862347263},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perceptron/perceptron-mk1-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.125e-08,"completion":3.4375000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.01081344,"max_completion_cost":0.022528000000000003,"max_cost":0.030638080000000005},"sats_pricing":{"prompt":6.342817036483653e-05,"completion":0.0005285680863736378,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":16.627314292119706,"max_completion_cost":34.64023810858273,"max_cost":47.11072382767251},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ring-2.6-1t-20260508","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":8.25e-07,"request":0.0,"image":1.375e-07,"web_search":0.007700000000000001,"internal_reasoning":8.25e-07,"input_cache_read":1.375e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.1441792,"max_completion_cost":0.0540672,"max_cost":0.18923520000000002},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0012685634072967305,"request":0.001,"image":0.0002114272345494551,"web_search":11.839925134769487,"internal_reasoning":0.0012685634072967305,"input_cache_read":2.114272345494551e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":221.69752389492942,"max_completion_cost":83.13657146059853,"max_cost":290.9780001120949},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite:batch","name":"Google: Gemini 3.1 Flash Lite (batch)","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":6.875e-08,"completion":4.125e-07,"request":0.0,"image":6.875e-08,"web_search":0.007700000000000001,"internal_reasoning":4.125e-07,"input_cache_read":6.875e-09,"input_cache_write":0.0,"max_prompt_cost":0.0720896,"max_completion_cost":0.0270336,"max_cost":0.09461760000000001},"sats_pricing":{"prompt":0.00010571361727472754,"completion":0.0006342817036483653,"request":0.001,"image":0.00010571361727472754,"web_search":11.839925134769487,"internal_reasoning":0.0006342817036483653,"input_cache_read":1.0571361727472754e-05,"input_cache_write":0.0,"max_prompt_cost":110.84876194746471,"max_completion_cost":41.568285730299266,"max_cost":145.48900005604744},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.3","name":"SpaceXAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.6875000000000001,"max_completion_cost":1.3750000000000002,"max_cost":1.3750000000000002},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":1057.1361727472756,"max_completion_cost":2114.2723454945512,"max_cost":2114.2723454945512},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.3-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.75e-08,"completion":5.5e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.00720896,"max_cost":0.00720896},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":8.457089381978204e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.228544690989102e-05,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":11.084876194746471,"max_cost":11.084876194746471},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.125e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2162688,"max_completion_cost":1.081344,"max_cost":1.081344},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.0063428170364836535,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":332.5462858423941,"max_completion_cost":1662.7314292119709,"max_cost":1662.7314292119709},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3.5-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-latest","name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":2.7500000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":6.875000000000001e-07,"max_prompt_cost":0.11,"max_completion_cost":0.17600000000000002,"max_cost":0.2508},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.004228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0010571361727472757,"max_prompt_cost":169.1417876395641,"max_completion_cost":270.6268602233025,"max_cost":385.6432758182061},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-haiku-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-mini-latest","name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":2.4750000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":4.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.165,"max_completion_cost":0.3168000000000001,"max_cost":0.4290000000000001},"sats_pricing":{"prompt":0.0006342817036483653,"completion":0.0038056902218901924,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":6.342817036483653e-05,"input_cache_write":0.0,"max_prompt_cost":253.71268145934613,"max_completion_cost":487.12834840194466,"max_cost":659.6529717943},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~openai/gpt-mini-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-pro-latest","name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":6.6e-06,"request":0.0,"image":1.1e-06,"web_search":0.007700000000000001,"internal_reasoning":6.6e-06,"input_cache_read":1.1e-07,"input_cache_write":2.0625e-07,"max_prompt_cost":1.1534336,"max_completion_cost":0.4325376,"max_cost":1.5138816000000002},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.010148507258373844,"request":0.001,"image":0.0016914178763956407,"web_search":11.839925134769487,"internal_reasoning":0.010148507258373844,"input_cache_read":0.00016914178763956407,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1773.5801911594353,"max_completion_cost":665.0925716847883,"max_cost":2327.824000896759},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~google/gemini-pro-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-latest","name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":7.7e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.595e-07,"input_cache_write":0.0,"max_prompt_cost":1.4417920000000002,"max_completion_cost":8.0740352,"max_cost":8.0740352},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.011839925134769487,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024525559207736787,"input_cache_write":0.0,"max_prompt_cost":2216.9752389492946,"max_completion_cost":12415.06133811605,"max_cost":12415.06133811605},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":1048576,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~moonshotai/kimi-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-flash-latest","name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.125e-06,"request":0.0,"image":8.25e-07,"web_search":0.007700000000000001,"internal_reasoning":4.125e-06,"input_cache_read":8.25e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.8650752,"max_completion_cost":0.270336,"max_cost":1.081344},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.0063428170364836535,"request":0.001,"image":0.0012685634072967305,"web_search":11.839925134769487,"internal_reasoning":0.0063428170364836535,"input_cache_read":0.00012685634072967305,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":1330.1851433695765,"max_completion_cost":415.6828573029927,"max_cost":1662.7314292119709},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~google/gemini-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":5.500000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":1.1,"max_completion_cost":0.7040000000000001,"max_cost":1.6632000000000002},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.008457089381978205,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0021142723454945513,"max_prompt_cost":1691.4178763956409,"max_completion_cost":1082.50744089321,"max_cost":2557.423829110209},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-sonnet-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-latest","name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.65e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.8875,"max_completion_cost":2.112,"max_cost":4.647500000000001},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.025371268145934614,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4439.971925538557,"max_completion_cost":3247.52232267963,"max_cost":7146.240527771583},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~openai/gpt-latest","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.0625e-07,"max_prompt_cost":0.165,"max_completion_cost":0.06488064,"max_cost":0.21906720000000002},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.00031714085182418263,"max_prompt_cost":253.71268145934613,"max_completion_cost":99.76388575271824,"max_cost":336.84925291994466},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-plus-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-flash","name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.03125e-07,"completion":6.187500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":1.2890625e-07,"max_prompt_cost":0.10312500000000001,"max_completion_cost":0.04055040000000001,"max_cost":0.136917},"sats_pricing":{"prompt":0.00015857042591209132,"completion":0.0009514225554725481,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.00019821303239011417,"max_prompt_cost":158.57042591209134,"max_completion_cost":62.35242859544891,"max_cost":210.53078307496543},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.1441792,"max_cost":0.1441792},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.228544690989102e-05,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":221.69752389492942,"max_cost":221.69752389492942},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-max-preview","name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":5.648500000000001e-07,"completion":3.3891e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":7.060625e-07,"max_prompt_cost":0.14807203840000002,"max_completion_cost":0.2221080576,"max_cost":0.3331620864},"sats_pricing":{"prompt":0.0008685430795291616,"completion":0.005211258477174969,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0010856788494114518,"max_prompt_cost":227.68335704009255,"max_completion_cost":341.52503556013875,"max_cost":512.2875533402082},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-max-preview-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-27b","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":1.98e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-08,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.51904512,"max_cost":0.51904512},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.0030445521775121533,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010148507258373844,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":798.1110860217459,"max_cost":798.1110860217459},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-27b-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.3925000000000003e-07,"completion":4.785000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.99375e-09,"input_cache_write":0.0,"max_prompt_cost":0.25087180800000003,"max_completion_cost":0.18374400000000002,"max_cost":0.34274380800000004},"sats_pricing":{"prompt":0.0003678833881160519,"completion":0.0007357667762321038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.065694900967099e-06,"input_cache_write":0.0,"max_prompt_cost":385.7536915771772,"max_completion_cost":282.5344420731279,"max_cost":527.0209126137412},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":7.700000000000001e-08,"completion":1.5400000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5400000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.08074035200000002,"max_completion_cost":0.06055526400000001,"max_cost":0.11101798400000001},"sats_pricing":{"prompt":0.00011839925134769487,"completion":0.00023679850269538974,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3679850269538972e-05,"input_cache_write":0.0,"max_prompt_cost":124.1506133811605,"max_completion_cost":93.11296003587037,"max_cost":170.70709339909567},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-flash-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.125e-08,"completion":3.4375000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.01081344,"max_completion_cost":0.011264000000000001,"max_cost":0.020725760000000003},"sats_pricing":{"prompt":6.342817036483653e-05,"completion":0.0005285680863736378,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":16.627314292119706,"max_completion_cost":17.320119054291364,"max_cost":31.869019059896107},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-2.6-1t-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.465e-08,"completion":1.1550000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1550000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.0090832896,"max_completion_cost":0.030277632000000002,"max_cost":0.030277632000000002},"sats_pricing":{"prompt":5.3279663106462686e-05,"completion":0.00017759887702154228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.775988770215423e-05,"input_cache_write":0.0,"max_prompt_cost":13.966944005380554,"max_completion_cost":46.55648001793518,"max_cost":46.55648001793518},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hy3-preview-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3925000000000003e-07,"completion":4.785000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.98e-09,"input_cache_write":0.0,"max_prompt_cost":0.25087180800000003,"max_completion_cost":0.06271795200000001,"max_cost":0.282230784},"sats_pricing":{"prompt":0.0003678833881160519,"completion":0.0007357667762321038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.0445521775121535e-06,"input_cache_write":0.0,"max_prompt_cost":385.7536915771772,"max_completion_cost":96.4384228942943,"max_cost":433.97290302432435},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.700000000000001e-08,"completion":1.5400000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.54e-09,"input_cache_write":0.0,"max_prompt_cost":0.08074035200000002,"max_completion_cost":0.020185088000000004,"max_cost":0.09083289600000001},"sats_pricing":{"prompt":0.00011839925134769487,"completion":0.00023679850269538974,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3679850269538973e-06,"input_cache_write":0.0,"max_prompt_cost":124.1506133811605,"max_completion_cost":31.037653345290124,"max_cost":139.66944005380554},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"xiaomi/mimo-v2.5-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":1.65e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.0014417920000000001,"max_completion_cost":0.000540672,"max_cost":0.00180224},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":2.537126814593461e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.691417876395641e-06,"input_cache_write":0.0,"max_prompt_cost":2.2169752389492943,"max_completion_cost":0.8313657146059853,"max_cost":2.7712190486866177},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-2.6-flash-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-latest","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-opus-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.18725e-07,"completion":1.342e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.368e-08,"input_cache_write":0.0,"max_prompt_cost":0.0835518464,"max_completion_cost":0.351797248,"max_cost":0.351797248},"sats_pricing":{"prompt":0.0004900883296856369,"completion":0.0020635298092026816,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.254119236810726e-05,"input_cache_write":0.0,"max_prompt_cost":128.4737150971116,"max_completion_cost":540.9419583036278,"max_cost":540.9419583036278},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.6-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7","name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7:batch","name":"Anthropic: Claude Opus 4.7 (batch)","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.7187500000000003e-06,"max_prompt_cost":1.3750000000000002,"max_completion_cost":0.8800000000000001,"max_cost":2.079},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0026428404318681892,"max_prompt_cost":2114.2723454945512,"max_completion_cost":1353.1343011165127,"max_cost":3196.7797863877613},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.236000000000001e-07,"completion":1.6456000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.724e-08,"input_cache_write":0.0,"max_prompt_cost":0.10616094720000001,"max_completion_cost":0.21569208320000002,"max_cost":0.2532237312},"sats_pricing":{"prompt":0.0008051149091643251,"completion":0.002530361143087879,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00014952134027337464,"input_cache_write":0.0,"max_prompt_cost":163.23865806288524,"max_completion_cost":331.65949574681446,"max_cost":389.3701324357133},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.1-20260406","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":3.850000000000001e-08,"completion":1.87e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.010092544000000002,"max_completion_cost":0.003063808,"max_cost":0.012525568},"sats_pricing":{"prompt":5.9199625673847435e-05,"completion":0.0002875410389872589,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":15.518826672645062,"max_completion_cost":4.71107238276725,"max_cost":19.259972388371995},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-31b-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.87e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.049020928,"max_cost":0.049020928},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0002875410389872589,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":75.377158124276,"max_cost":75.377158124276},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-4-31b-it-20260402","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-plus","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.7875e-07,"completion":1.0725000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.234375e-07,"max_prompt_cost":0.17875000000000002,"max_completion_cost":0.07028736000000001,"max_cost":0.2373228},"sats_pricing":{"prompt":0.0002748554049142916,"completion":0.00164913242948575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0003435692561428645,"max_prompt_cost":274.8554049142916,"max_completion_cost":108.0775428987781,"max_cost":364.9200239966067},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-plus-04-02","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.6e-07,"completion":2.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.32e-07,"input_cache_write":0.0,"max_prompt_cost":0.13381632000000002,"max_completion_cost":0.2883584,"max_cost":0.3356672},"sats_pricing":{"prompt":0.0010148507258373844,"completion":0.0033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00020297014516747688,"input_cache_write":0.0,"max_prompt_cost":205.7630143649814,"max_completion_cost":443.39504778985884,"max_cost":516.1395478178825},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5v-turbo-20260401","alias_ids":null,"forwarded_model_id":null},{"id":"trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.21e-07,"completion":4.675e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.031719424,"max_completion_cost":0.12255232,"max_cost":0.12255232},"sats_pricing":{"prompt":0.0001860559664035205,"completion":0.0007188525974681473,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.074253629186922e-05,"input_cache_write":0.0,"max_prompt_cost":48.773455256884475,"max_completion_cost":188.44289531069,"max_cost":188.44289531069},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"arcee-ai/trinity-large-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":1.3750000000000002,"max_completion_cost":2.7500000000000004,"max_cost":2.7500000000000004},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":2114.2723454945512,"max_completion_cost":4228.5446909891025,"max_cost":4228.5446909891025},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20","name":"SpaceXAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":1.3750000000000002,"max_completion_cost":2.7500000000000004,"max_cost":2.7500000000000004},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":2114.2723454945512,"max_completion_cost":4228.5446909891025,"max_cost":4228.5446909891025},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.20-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.04224,"max_completion_cost":0.0528,"max_cost":0.08184},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.074253629186922e-05,"input_cache_write":0.0,"max_prompt_cost":64.9504464535926,"max_completion_cost":81.18805806699075,"max_cost":125.84149000383566},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","alias_ids":null,"forwarded_model_id":null},{"id":"reka-edge","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":5.5e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00090112,"max_completion_cost":0.00090112,"max_cost":0.00090112},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":8.457089381978204e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3856095243433089,"max_completion_cost":1.3856095243433089,"max_cost":1.3856095243433089},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"rekaai/reka-edge-2603","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.033792,"max_completion_cost":0.08650752,"max_cost":0.09867264},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.074253629186922e-05,"input_cache_write":0.0,"max_prompt_cost":51.96035716287409,"max_completion_cost":133.01851433695765,"max_cost":151.72424291559233},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.7-20260318","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.08650752,"max_cost":0.08650752},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":133.01851433695765,"max_cost":133.01851433695765},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-2603","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-turbo","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.6e-07,"completion":2.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.32e-07,"input_cache_write":0.0,"max_prompt_cost":0.13381632000000002,"max_completion_cost":0.2883584,"max_cost":0.3356672},"sats_pricing":{"prompt":0.0010148507258373844,"completion":0.0033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00020297014516747688,"input_cache_write":0.0,"max_prompt_cost":205.7630143649814,"max_completion_cost":443.39504778985884,"max_cost":516.1395478178825},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5-turbo-20260315","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.675e-08,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012255232,"max_completion_cost":0.00360448,"max_cost":0.01509376},"sats_pricing":{"prompt":7.188525974681472e-05,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.844289531069,"max_completion_cost":5.5424380973732355,"max_cost":23.208959532750423},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0360448,"max_completion_cost":0.1441792,"max_cost":0.1622016},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55.424380973732355,"max_completion_cost":221.69752389492942,"max_cost":249.4097143817956},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-9b","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":8.25e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.02162688,"max_cost":0.02162688},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":33.25462858423941,"max_cost":33.25462858423941},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-9b-20260310","alias_ids":null,"forwarded_model_id":null},{"id":"mercury-2","name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":4.125e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.0176,"max_completion_cost":0.020625,"max_cost":0.03135},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0006342817036483653,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.114272345494551e-05,"input_cache_write":0.0,"max_prompt_cost":27.062686022330254,"max_completion_cost":31.714085182418266,"max_cost":48.205409477275765},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inception/mercury-2-20260304","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":8.25e-07,"request":0.0,"image":1.375e-07,"web_search":0.007700000000000001,"internal_reasoning":8.25e-07,"input_cache_read":1.375e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.1441792,"max_completion_cost":0.0540672,"max_cost":0.18923520000000002},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0012685634072967305,"request":0.001,"image":0.0002114272345494551,"web_search":11.839925134769487,"internal_reasoning":0.0012685634072967305,"input_cache_read":2.114272345494551e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":221.69752389492942,"max_completion_cost":83.13657146059853,"max_cost":290.9780001120949},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.02883584,"max_cost":0.0360448},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":44.339504778985884,"max_cost":55.424380973732355},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.007700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0180224,"max_completion_cost":0.1081344,"max_cost":0.1081344},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.002537126814593461,"request":0.001,"image":0.0,"web_search":11.839925134769487,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.712190486866177,"max_completion_cost":166.27314292119706,"max_cost":166.27314292119706},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.700000000000001e-08,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.020185088000000004,"max_completion_cost":0.1441792,"max_cost":0.1441792},"sats_pricing":{"prompt":0.00011839925134769487,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":31.037653345290124,"max_completion_cost":221.69752389492942,"max_cost":221.69752389492942},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-27b","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.0725000000000001e-07,"completion":8.580000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.028114944000000003,"max_completion_cost":0.056229888000000006,"max_cost":0.077316096},"sats_pricing":{"prompt":0.00016491324294857498,"completion":0.0013193059435885998,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.23101715951124,"max_completion_cost":86.46203431902248,"max_cost":118.8852971886559},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-27b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.595e-07,"completion":1.32e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.041811968,"max_completion_cost":0.1081344,"max_cost":0.136880128},"sats_pricing":{"prompt":0.00024525559207736787,"completion":0.002029701451674769,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":64.29228192952952,"max_completion_cost":166.27314292119706,"max_cost":210.4740867477486},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.5750000000000006e-08,"completion":1.4300000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.035750000000000004,"max_completion_cost":0.009371648000000002,"max_cost":0.04277873600000001},"sats_pricing":{"prompt":5.497108098285833e-05,"completion":0.00021988432393143332,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":54.97108098285833,"max_completion_cost":14.410339053170414,"max_cost":65.77883527273615},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-flash-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":6.6e-06,"request":0.0,"image":1.1e-06,"web_search":0.007700000000000001,"internal_reasoning":6.6e-06,"input_cache_read":1.1e-07,"input_cache_write":2.0625e-07,"max_prompt_cost":1.1534336,"max_completion_cost":0.4325376,"max_cost":1.5138816000000002},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.010148507258373844,"request":0.001,"image":0.0016914178763956407,"web_search":11.839925134769487,"internal_reasoning":0.010148507258373844,"input_cache_read":0.00016914178763956407,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1773.5801911594353,"max_completion_cost":665.0925716847883,"max_cost":2327.824000896759},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"aion-2.0","name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.4e-07,"completion":8.8e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.05767168,"max_completion_cost":0.02883584,"max_cost":0.0720896},"sats_pricing":{"prompt":0.0006765671505582563,"completion":0.0013531343011165126,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":88.67900955797177,"max_completion_cost":44.339504778985884,"max_cost":110.84876194746471},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-2.0-20260223","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":6.6e-06,"request":0.0,"image":1.1e-06,"web_search":0.007700000000000001,"internal_reasoning":6.6e-06,"input_cache_read":1.1e-07,"input_cache_write":2.0625e-07,"max_prompt_cost":1.1534336,"max_completion_cost":0.4325376,"max_cost":1.5138816000000002},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.010148507258373844,"request":0.001,"image":0.0016914178763956407,"web_search":11.839925134769487,"internal_reasoning":0.010148507258373844,"input_cache_read":0.00016914178763956407,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1773.5801911594353,"max_completion_cost":665.0925716847883,"max_cost":2327.824000896759},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview:batch","name":"Google: Gemini 3.1 Pro Preview (batch)","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":3.3e-06,"request":0.0,"image":5.5e-07,"web_search":0.007700000000000001,"internal_reasoning":3.3e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5767168,"max_completion_cost":0.2162688,"max_cost":0.7569408000000001},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.005074253629186922,"request":0.001,"image":0.0008457089381978204,"web_search":11.839925134769487,"internal_reasoning":0.005074253629186922,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":886.7900955797177,"max_completion_cost":332.5462858423941,"max_cost":1163.9120004483796},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":2.0625e-06,"max_prompt_cost":1.6500000000000001,"max_completion_cost":1.056,"max_cost":2.4948},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0031714085182418267,"max_prompt_cost":2537.1268145934614,"max_completion_cost":1623.761161339815,"max_cost":3836.1357436653134},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6:batch","name":"Anthropic: Claude Sonnet 4.6 (batch)","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.125e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":1.03125e-06,"max_prompt_cost":0.8250000000000001,"max_completion_cost":0.528,"max_cost":1.2474},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.0063428170364836535,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0015857042591209134,"max_prompt_cost":1268.5634072967307,"max_completion_cost":811.8805806699075,"max_cost":1918.0678718326567},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.4300000000000002e-07,"completion":8.580000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14300000000000002,"max_completion_cost":0.056229888000000006,"max_cost":0.18985824000000004},"sats_pricing":{"prompt":0.00021988432393143332,"completion":0.0013193059435885998,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":219.88432393143333,"max_completion_cost":86.46203431902248,"max_cost":291.9360191972854},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-plus-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1450000000000002e-07,"completion":1.2870000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.056229888000000006,"max_completion_cost":0.08434483200000001,"max_cost":0.12651724800000003},"sats_pricing":{"prompt":0.00032982648589714996,"completion":0.0019789589153829,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":86.46203431902248,"max_completion_cost":129.69305147853373,"max_cost":194.5395772178006},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.21e-07,"completion":4.95e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.023789568000000004,"max_completion_cost":0.09732096000000001,"max_cost":0.09732096000000001},"sats_pricing":{"prompt":0.0001860559664035205,"completion":0.0007611380443780383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.228544690989102e-05,"input_cache_write":0.00031714085182418263,"max_prompt_cost":36.58009144266336,"max_completion_cost":149.64582862907739,"max_cost":149.64582862907739},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.225000000000001e-07,"completion":1.4025000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.10700800000000002,"max_completion_cost":0.18382848000000002,"max_cost":0.22235136000000003},"sats_pricing":{"prompt":0.0008034234912879294,"completion":0.002156557792404442,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":164.54113101576797,"max_completion_cost":282.664342966035,"max_cost":341.8991501317115},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":4.2900000000000004e-07,"completion":2.1450000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11245977600000001,"max_completion_cost":0.14057472000000001,"max_cost":0.22491955200000002},"sats_pricing":{"prompt":0.0006596529717942999,"completion":0.0032982648589715,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":172.92406863804496,"max_completion_cost":216.1550857975562,"max_cost":345.8481372760899},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-max-thinking-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":2.7500000000000004,"max_completion_cost":1.7600000000000002,"max_cost":4.158},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":4228.5446909891025,"max_completion_cost":2706.2686022330254,"max_cost":6393.559572775523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6:batch","name":"Anthropic: Claude Opus 4.6 (batch)","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.7187500000000003e-06,"max_prompt_cost":1.3750000000000002,"max_completion_cost":0.8800000000000001,"max_cost":2.079},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0026428404318681892,"max_prompt_cost":2114.2723454945512,"max_completion_cost":1353.1343011165127,"max_cost":3196.7797863877613},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":6.6e-08,"completion":4.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.850000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.017301504,"max_completion_cost":0.11534336,"max_cost":0.11534336},"sats_pricing":{"prompt":0.00010148507258373844,"completion":0.0006765671505582563,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.9199625673847435e-05,"input_cache_write":0.0,"max_prompt_cost":26.60370286739153,"max_completion_cost":177.35801911594353,"max_cost":177.35801911594353},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.5-flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.65e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.01081344,"max_cost":0.02162688},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":16.627314292119706,"max_cost":33.25462858423941},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"stepfun/step-3.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1350000000000005e-07,"completion":1.5675e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.225000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.08218214400000001,"max_completion_cost":0.41091072,"max_cost":0.41091072},"sats_pricing":{"prompt":0.00048205409477275766,"completion":0.002410270473863788,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.034234912879294e-05,"input_cache_write":0.0,"max_prompt_cost":126.36758862010979,"max_completion_cost":631.8379431005488,"max_cost":631.8379431005488},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.5-0127","alias_ids":null,"forwarded_model_id":null},{"id":"solar-pro-3","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.01081344,"max_completion_cost":0.04325376,"max_cost":0.04325376},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":16.627314292119706,"max_completion_cost":66.50925716847883,"max_cost":66.50925716847883},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"upstage/solar-pro-3","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2-her","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.01081344,"max_completion_cost":0.00135168,"max_cost":0.0118272},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":16.627314292119706,"max_completion_cost":2.0784142865149633,"max_cost":18.186125007005927},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2-her-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"palmyra-x5","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.3432,"max_completion_cost":0.0270336,"max_cost":0.36753024},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":527.72237743544,"max_completion_cost":41.568285730299266,"max_cost":565.1338345927093},"per_request_limits":null,"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"writer/palmyra-x5-20250428","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7-flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-08,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5000000000000004e-09,"input_cache_write":0.0,"max_prompt_cost":0.006690816,"max_completion_cost":0.00360448,"max_cost":0.009754624},"sats_pricing":{"prompt":5.074253629186922e-05,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-06,"input_cache_write":0.0,"max_prompt_cost":10.288150718249067,"max_completion_cost":5.5424380973732355,"max_cost":14.999223101016318},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.7-flash-20260119","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.125e-08,"completion":1.65e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01081344,"max_completion_cost":0.00540672,"max_cost":0.01486848},"sats_pricing":{"prompt":6.342817036483653e-05,"completion":0.0002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.627314292119706,"max_completion_cost":8.313657146059853,"max_cost":22.862557151664596},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0360448,"max_completion_cost":0.0360448,"max_cost":0.067584},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55.424380973732355,"max_completion_cost":55.424380973732355,"max_cost":103.92071432574818},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-1.6-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.033792,"max_completion_cost":0.08650752,"max_cost":0.09867264},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.00031714085182418263,"max_prompt_cost":51.96035716287409,"max_completion_cost":133.01851433695765,"max_cost":151.72424291559233},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2e-07,"completion":9.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4000000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.04460544,"max_completion_cost":0.1261568,"max_cost":0.1419264},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.0014799906418461858,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.765671505582563e-05,"input_cache_write":0.0,"max_prompt_cost":68.5876714549938,"max_completion_cost":193.98533340806327,"max_cost":218.23350008407115},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.7-20251222","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.65e-06,"request":0.0,"image":2.75e-07,"web_search":0.007700000000000001,"internal_reasoning":1.65e-06,"input_cache_read":2.75e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.2883584,"max_completion_cost":0.1081344,"max_cost":0.37847040000000004},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.002537126814593461,"request":0.001,"image":0.0004228544690989102,"web_search":11.839925134769487,"internal_reasoning":0.002537126814593461,"input_cache_read":4.228544690989102e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":443.39504778985884,"max_completion_cost":166.27314292119706,"max_cost":581.9560002241898},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview:batch","name":"Google: Gemini 3 Flash Preview (batch)","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":8.25e-07,"request":0.0,"image":1.375e-07,"web_search":0.007700000000000001,"internal_reasoning":8.25e-07,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1441792,"max_completion_cost":0.0540672,"max_cost":0.18923520000000002},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0012685634072967305,"request":0.001,"image":0.0002114272345494551,"web_search":11.839925134769487,"internal_reasoning":0.0012685634072967305,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":221.69752389492942,"max_completion_cost":83.13657146059853,"max_cost":290.9780001120949},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.75e-08,"completion":1.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.00720896,"max_completion_cost":0.02883584,"max_cost":0.02883584},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":0.00016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":11.084876194746471,"max_completion_cost":44.339504778985884,"max_cost":44.339504778985884},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","alias_ids":null,"forwarded_model_id":null},{"id":"relace-search","name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1408,"max_completion_cost":0.2112,"max_cost":0.2816},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":216.50148817864203,"max_completion_cost":324.752232267963,"max_cost":433.00297635728407},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"relace/relace-search-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6v","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":4.95e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.025e-08,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.01622016,"max_cost":0.03244032},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0007611380443780383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.651399160088012e-05,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":24.94097143817956,"max_cost":49.88194287635912},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.6-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"nova-2-lite-v1","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.165,"max_completion_cost":0.09011062500000001,"max_cost":0.24429735000000002},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":253.71268145934613,"max_completion_cost":138.55883816198542,"max_cost":375.6444590418933},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-2-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":1.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.02883584,"max_cost":0.02883584},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.00016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6914178763956408e-05,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":44.339504778985884,"max_cost":44.339504778985884},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-14b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":8.25e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-09,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.02162688,"max_cost":0.02162688},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.00012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2685634072967305e-05,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":33.25462858423941,"max_cost":33.25462858423941},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-8b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":5.5e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5000000000000004e-09,"input_cache_write":0.0,"max_prompt_cost":0.00720896,"max_completion_cost":0.00720896,"max_cost":0.00720896},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":8.457089381978204e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-06,"input_cache_write":0.0,"max_prompt_cost":11.084876194746471,"max_completion_cost":11.084876194746471,"max_cost":11.084876194746471},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-3b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2512","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":8.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.0720896,"max_completion_cost":0.2162688,"max_cost":0.2162688},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.0012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.228544690989102e-05,"input_cache_write":0.0,"max_prompt_cost":110.84876194746471,"max_completion_cost":332.5462858423941,"max_cost":332.5462858423941},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large-2512","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.4795e-07,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.3975e-08,"input_cache_write":0.0,"max_prompt_cost":0.024240128,"max_completion_cost":0.01441792,"max_cost":0.0289619968},"sats_pricing":{"prompt":0.00022749570437521368,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011374785218760684,"input_cache_write":0.0,"max_prompt_cost":37.27289620483501,"max_completion_cost":22.169752389492942,"max_cost":44.53349011239395},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.2-20251201","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7500000000000004e-06,"completion":1.3750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":3.4375000000000005e-06,"max_prompt_cost":0.55,"max_completion_cost":0.8800000000000001,"max_cost":1.2540000000000002},"sats_pricing":{"prompt":0.004228544690989103,"completion":0.021142723454945514,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0004228544690989102,"input_cache_write":0.0052856808637363785,"max_prompt_cost":845.7089381978204,"max_completion_cost":1353.1343011165127,"max_cost":1928.2163790910308},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5:batch","name":"Anthropic: Claude Opus 4.5 (batch)","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":1.7187500000000003e-06,"max_prompt_cost":0.275,"max_completion_cost":0.44000000000000006,"max_cost":0.6270000000000001},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0026428404318681892,"max_prompt_cost":422.8544690989102,"max_completion_cost":676.5671505582563,"max_cost":964.1081895455154},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":2.75e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00540672,"max_completion_cost":0.0180224,"max_cost":0.0180224},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0004228544690989102,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.313657146059853,"max_completion_cost":27.712190486866177,"max_cost":27.712190486866177},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"allenai/olmo-3-32b-think-20251121","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":6.6e-06,"request":0.0,"image":1.1e-06,"web_search":0.007700000000000001,"internal_reasoning":6.6e-06,"input_cache_read":1.1e-07,"input_cache_write":2.0625e-07,"max_prompt_cost":0.0720896,"max_completion_cost":0.2162688,"max_cost":0.2523136},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.010148507258373844,"request":0.001,"image":0.0016914178763956407,"web_search":11.839925134769487,"internal_reasoning":0.010148507258373844,"input_cache_read":0.00016914178763956407,"input_cache_write":0.00031714085182418263,"max_prompt_cost":110.84876194746471,"max_completion_cost":332.5462858423941,"max_cost":387.97066681612654},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-pro-image-preview-20251120","alias_ids":null,"forwarded_model_id":null},{"id":"cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":6.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08800000000000001,"max_completion_cost":0.08800000000000001,"max_cost":0.08800000000000001},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.0010571361727472757,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":135.31343011165126,"max_completion_cost":135.31343011165126,"max_cost":135.31343011165126},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.13798400000000002,"max_cost":0.19137536000000002},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":212.1714584150692,"max_cost":294.26882273241023},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","alias_ids":null,"forwarded_model_id":null},{"id":"nova-premier-v1","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":6.875000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.4375000000000004e-07,"input_cache_write":0.0,"max_prompt_cost":1.3750000000000002,"max_completion_cost":0.22000000000000003,"max_cost":1.5510000000000002},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.010571361727472757,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005285680863736378,"input_cache_write":0.0,"max_prompt_cost":2114.2723454945512,"max_completion_cost":338.2835752791282,"max_cost":2384.899205717854},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-premier-v1","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro-search","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0099,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.33,"max_completion_cost":0.066,"max_cost":0.38280000000000003},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":15.222760887560767,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":507.42536291869226,"max_completion_cost":101.48507258373844,"max_cost":588.613420985683},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-pro-search","alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.65e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5000000000000004e-09,"input_cache_write":0.0,"max_prompt_cost":0.00176,"max_completion_cost":0.00528,"max_cost":0.00528},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-06,"input_cache_write":0.0,"max_prompt_cost":2.7062686022330253,"max_completion_cost":8.118805806699076,"max_cost":8.118805806699076},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/voxtral-small-24b-2507","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4025e-07,"completion":5.61e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.0287232,"max_completion_cost":0.073531392,"max_cost":0.083871744},"sats_pricing":{"prompt":0.0002156557792404442,"completion":0.0008626231169617768,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.00031714085182418263,"max_prompt_cost":44.16630358844297,"max_completion_cost":113.06573718641401,"max_cost":128.96560647825348},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":5.720000000000001e-08,"completion":2.2880000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007497318400000001,"max_completion_cost":0.007497318400000001,"max_cost":0.013120307200000002},"sats_pricing":{"prompt":8.795372957257334e-05,"completion":0.00035181491829029335,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.528271242536333,"max_completion_cost":11.528271242536333,"max_cost":20.17447467443858},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.350000000000001e-09,"completion":6.160000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0012248500000000002,"max_completion_cost":0.008069600000000001,"max_cost":0.008069600000000001},"sats_pricing":{"prompt":1.4377051949362948e-05,"completion":9.471940107815589e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.8833938053665462,"max_completion_cost":12.408241541238423,"max_cost":12.408241541238423},"per_request_limits":null,"top_provider":{"context_length":131000,"max_completion_tokens":131000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ibm-granite/granite-4.0-h-micro","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":2.7500000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":6.875000000000001e-07,"max_prompt_cost":0.11,"max_completion_cost":0.17600000000000002,"max_cost":0.2508},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.004228544690989103,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0010571361727472757,"max_prompt_cost":169.1417876395641,"max_completion_cost":270.6268602233025,"max_cost":385.6432758182061},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5:batch","name":"Anthropic: Claude Haiku 4.5 (batch)","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":3.4375000000000004e-07,"max_prompt_cost":0.055,"max_completion_cost":0.08800000000000001,"max_cost":0.1254},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":4.228544690989102e-05,"input_cache_write":0.0005285680863736378,"max_prompt_cost":84.57089381978204,"max_completion_cost":135.31343011165126,"max_cost":192.82163790910306},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.900000000000001e-08,"completion":1.155e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012976128000000002,"max_completion_cost":0.03784704,"max_cost":0.047579136},"sats_pricing":{"prompt":0.00015222760887560768,"completion":0.0017759887702154227,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.95277715054365,"max_completion_cost":58.19560002241897,"max_cost":73.1601828853267},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-8b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.435e-08,"completion":2.5025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0084344832,"max_completion_cost":0.008200192,"max_cost":0.0145260544},"sats_pricing":{"prompt":9.894794576914499e-05,"completion":0.00038479756688000825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":12.969305147853373,"max_completion_cost":12.60904667152411,"max_cost":22.33602553241414},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":1.65e-07,"web_search":0.007700000000000001,"internal_reasoning":1.3750000000000002e-06,"input_cache_read":1.65e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.00540672,"max_completion_cost":0.011264000000000001,"max_cost":0.015319040000000003},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0021142723454945513,"request":0.001,"image":0.0002537126814593461,"web_search":11.839925134769487,"internal_reasoning":0.0021142723454945513,"input_cache_read":2.537126814593461e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":8.313657146059853,"max_completion_cost":17.320119054291364,"max_cost":23.555361913836254},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-image","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":1.32e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.04325376,"max_cost":0.0540672},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.002029701451674769,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":66.50925716847883,"max_cost":83.13657146059853},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.00540672,"max_cost":0.02568192},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":8.313657146059853,"max_cost":39.4898714437843},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.75e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.0557568,"max_completion_cost":0.1441792,"max_cost":0.16389120000000001},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":85.73458931874224,"max_completion_cost":221.69752389492942,"max_cost":252.00773223993932},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.6","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":2.0625e-06,"max_prompt_cost":1.6500000000000001,"max_completion_cost":0.528,"max_cost":2.0724},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0031714085182418267,"max_prompt_cost":2537.1268145934614,"max_completion_cost":811.8805806699075,"max_cost":3186.631279129387},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5:batch","name":"Anthropic: Claude Sonnet 4.5 (batch)","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":8.25e-07,"completion":4.125e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":1.03125e-06,"max_prompt_cost":0.8250000000000001,"max_completion_cost":0.264,"max_cost":1.0362},"sats_pricing":{"prompt":0.0012685634072967305,"completion":0.0063428170364836535,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0015857042591209134,"max_prompt_cost":1268.5634072967307,"max_completion_cost":405.9402903349538,"max_cost":1593.3156395646936},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":1.485e-07,"completion":2.255e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024330240000000003,"max_completion_cost":0.014778368,"max_cost":0.029376512},"sats_pricing":{"prompt":0.0002283414133134115,"completion":0.0003467406646611063,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":37.411457157269346,"max_completion_cost":22.723996199230264,"max_cost":45.17087049359187},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.2-exp","alias_ids":null,"forwarded_model_id":null},{"id":"cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":2.75e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.0360448,"max_cost":0.0360448},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0004228544690989102,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012685634072967305,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":55.424380973732355,"max_cost":55.424380973732355},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/cydonia-24b-v4.1","alias_ids":null,"forwarded_model_id":null},{"id":"relace-apply-3","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.675e-07,"completion":6.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11968000000000001,"max_completion_cost":0.08800000000000001,"max_cost":0.14784000000000003},"sats_pricing":{"prompt":0.0007188525974681473,"completion":0.0010571361727472757,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":184.02626495184572,"max_completion_cost":135.31343011165126,"max_cost":227.32656258757416},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"relace/relace-apply-3","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2e-07,"completion":2.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.0720896,"max_cost":0.09371648},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.0033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":110.84876194746471,"max_cost":144.10339053170412},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.1550000000000001e-07,"completion":1.0450000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.015138816000000001,"max_completion_cost":0.034242560000000005,"max_cost":0.045596672000000005},"sats_pricing":{"prompt":0.00017759887702154228,"completion":0.0016068469825758589,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":23.27824000896759,"max_completion_cost":52.653161925045744,"max_cost":70.11184193177144},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.2900000000000004e-07,"completion":2.1450000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.58e-08,"input_cache_write":5.362500000000001e-07,"max_prompt_cost":0.11245977600000001,"max_completion_cost":0.14057472000000001,"max_cost":0.22491955200000002},"sats_pricing":{"prompt":0.0006596529717942999,"completion":0.0032982648589715,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00013193059435885997,"input_cache_write":0.000824566214742875,"max_prompt_cost":172.92406863804496,"max_completion_cost":216.1550857975562,"max_cost":345.8481372760899},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-max","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.575e-07,"completion":1.7875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.150000000000001e-08,"input_cache_write":4.46875e-07,"max_prompt_cost":0.35750000000000004,"max_completion_cost":0.1171456,"max_cost":0.45121648000000003},"sats_pricing":{"prompt":0.0005497108098285832,"completion":0.002748554049142916,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010994216196571666,"input_cache_write":0.000687138512285729,"max_prompt_cost":549.7108098285833,"max_completion_cost":180.12923816463015,"max_cost":693.8142003602874},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-plus","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":1.485e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.425e-08,"input_cache_write":0.0,"max_prompt_cost":0.019464192,"max_completion_cost":0.0180224,"max_cost":0.032620544},"sats_pricing":{"prompt":0.0002283414133134115,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011417070665670575,"input_cache_write":0.0,"max_prompt_cost":29.929165725815473,"max_completion_cost":27.712190486866177,"max_cost":50.159064781227784},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.1-terminus","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.0725000000000001e-07,"completion":5.362500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.145e-08,"input_cache_write":1.3406250000000001e-07,"max_prompt_cost":0.10725000000000001,"max_completion_cost":0.035143680000000004,"max_cost":0.13536494400000001},"sats_pricing":{"prompt":0.00016491324294857498,"completion":0.000824566214742875,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.298264858971499e-05,"input_cache_write":0.00020614155368571874,"max_prompt_cost":164.91324294857498,"max_completion_cost":54.03877144938905,"max_cost":208.14426010808623},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.17301504,"max_cost":0.17301504},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":266.0370286739153,"max_cost":266.0370286739153},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.9500000000000006e-08,"completion":6.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012976128000000002,"max_completion_cost":0.00991232,"max_cost":0.022077440000000004},"sats_pricing":{"prompt":7.611380443780384e-05,"completion":0.0009302798320176025,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.95277715054365,"max_completion_cost":15.2417047677764,"max_cost":33.947433346411074},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.4300000000000002e-07,"completion":4.2900000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14300000000000002,"max_completion_cost":0.014057472000000001,"max_cost":0.15237164800000003},"sats_pricing":{"prompt":0.00021988432393143332,"completion":0.0006596529717942999,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":219.88432393143333,"max_completion_cost":21.61550857975562,"max_cost":234.29466298460375},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.75e-07,"max_prompt_cost":0.22,"max_completion_cost":0.02162688,"max_cost":0.23441792},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0004228544690989102,"max_prompt_cost":338.2835752791282,"max_completion_cost":33.25462858423941,"max_cost":360.4533276686211},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.13798400000000002,"max_cost":0.19137536000000002},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":212.1714584150692,"max_cost":294.26882273241023},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2-0905","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":1.32e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0090112,"max_completion_cost":0.04325376,"max_cost":0.048660480000000006},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.002029701451674769,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.856095243433089,"max_completion_cost":66.50925716847883,"max_cost":74.82291431453869},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-70b","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":7.150000000000001e-08,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.009371648000000002,"max_completion_cost":0.02883584,"max_cost":0.02883584},"sats_pricing":{"prompt":0.00010994216196571666,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14.410339053170414,"max_completion_cost":44.339504778985884,"max_cost":44.339504778985884},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-4-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-405b","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0720896,"max_completion_cost":0.2162688,"max_cost":0.2162688},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":110.84876194746471,"max_completion_cost":332.5462858423941,"max_cost":332.5462858423941},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-4-405b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":1.375e-07,"completion":5.225000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.150000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.022528,"max_completion_cost":0.017121280000000003,"max_cost":0.035143680000000004},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0008034234912879294,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010994216196571666,"input_cache_write":0.0,"max_prompt_cost":34.64023810858272,"max_completion_cost":26.326580962522872,"max_cost":54.03877144938905},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3.1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.1441792,"max_cost":0.1441792},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3828357527912815e-05,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":221.69752389492942,"max_cost":221.69752389492942},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5v","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":9.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.05e-08,"input_cache_write":0.0,"max_prompt_cost":0.02162688,"max_completion_cost":0.01622016,"max_cost":0.03244032},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.0015222760887560766,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.302798320176024e-05,"input_cache_write":0.0,"max_prompt_cost":33.25462858423941,"max_completion_cost":24.94097143817956,"max_cost":49.88194287635912},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5v","alias_ids":null,"forwarded_model_id":null},{"id":"jamba-large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":4.4e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2816,"max_completion_cost":0.0180224,"max_cost":0.2951168},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.006765671505582563,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":433.00297635728407,"max_completion_cost":27.712190486866177,"max_cost":453.78711922243366},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ai21/jamba-large-1.7","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":8.25e-06,"completion":4.125e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":8.25e-07,"input_cache_write":1.03125e-05,"max_prompt_cost":1.6500000000000001,"max_completion_cost":1.32,"max_cost":2.7060000000000004},"sats_pricing":{"prompt":0.012685634072967307,"completion":0.06342817036483653,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0012685634072967305,"input_cache_write":0.015857042591209132,"max_prompt_cost":2537.1268145934614,"max_completion_cost":2029.701451674769,"max_cost":4160.887975933277},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1:batch","name":"Anthropic: Claude Opus 4.1 (batch)","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":4.125e-06,"completion":2.0625e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":4.125e-07,"input_cache_write":5.15625e-06,"max_prompt_cost":0.8250000000000001,"max_completion_cost":0.66,"max_cost":1.3530000000000002},"sats_pricing":{"prompt":0.0063428170364836535,"completion":0.031714085182418264,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0006342817036483653,"input_cache_write":0.007928521295604566,"max_prompt_cost":1268.5634072967307,"max_completion_cost":1014.8507258373845,"max_cost":2080.4439879666384},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-2508","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":4.95e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.04224,"max_completion_cost":0.12672,"max_cost":0.12672},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0007611380443780383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":64.9504464535926,"max_completion_cost":194.8513393607778,"max_cost":194.8513393607778},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/codestral-2508","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.850000000000001e-08,"completion":1.485e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.006160000000000001,"max_completion_cost":0.004866048,"max_cost":0.009764480000000002},"sats_pricing":{"prompt":5.9199625673847435e-05,"completion":0.0002283414133134115,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.47194010781559,"max_completion_cost":7.482291431453868,"max_cost":15.014378205188827},"per_request_limits":null,"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.6482500000000002e-08,"completion":1.0617750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0033897600000000003,"max_completion_cost":0.0033976800000000006,"max_cost":0.005940000000000001},"sats_pricing":{"prompt":4.0720885374225056e-05,"completion":0.00016326411051908923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.212273327900807,"max_completion_cost":5.224451536610856,"max_cost":9.133656532536461},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-07,"completion":1.21e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.05e-08,"input_cache_write":0.0,"max_prompt_cost":0.04325376,"max_completion_cost":0.11894784,"max_cost":0.12976128},"sats_pricing":{"prompt":0.0005074253629186922,"completion":0.001860559664035205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.302798320176024e-05,"input_cache_write":0.0,"max_prompt_cost":66.50925716847883,"max_completion_cost":182.90045721331677,"max_cost":199.52777150543648},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5-air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.150000000000001e-08,"completion":4.675e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.009371648000000002,"max_completion_cost":0.045957120000000004,"max_cost":0.04830003200000001},"sats_pricing":{"prompt":0.00010994216196571666,"completion":0.0007188525974681473,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.114272345494551e-05,"input_cache_write":0.0,"max_prompt_cost":14.410339053170414,"max_completion_cost":70.66608574150875,"max_cost":74.26867050480136},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5-air","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.265e-07,"completion":1.2650000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.016580608,"max_completion_cost":0.16580608000000002,"max_cost":0.16580608000000002},"sats_pricing":{"prompt":0.0001945130557854987,"completion":0.001945130557854987,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25.495215247916885,"max_completion_cost":254.95215247916886,"max_cost":254.95215247916886},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.04325376,"max_completion_cost":0.0360448,"max_cost":0.06848512000000001},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":66.50925716847883,"max_completion_cost":55.424380973732355,"max_cost":105.30632385009149},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.00704,"max_completion_cost":0.00022528,"max_cost":0.00715264},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.457089381978204e-05,"input_cache_write":0.0,"max_prompt_cost":10.825074408932101,"max_completion_cost":0.3464023810858272,"max_cost":10.998275599475015},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance/ui-tars-1.5-7b","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":2.2e-07,"request":0.0,"image":5.5e-08,"web_search":0.007700000000000001,"internal_reasoning":2.2e-07,"input_cache_read":5.5000000000000004e-09,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.05767168,"max_completion_cost":0.0144177,"max_cost":0.068484955},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00033828357527912814,"request":0.001,"image":8.457089381978204e-05,"web_search":11.839925134769487,"internal_reasoning":0.00033828357527912814,"input_cache_read":8.457089381978204e-06,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":88.67900955797177,"max_completion_cost":22.169414105917664,"max_cost":105.30607013741002},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite:batch","name":"Google: Gemini 2.5 Flash Lite (batch)","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.75e-08,"completion":1.1e-07,"request":0.0,"image":2.75e-08,"web_search":0.007700000000000001,"internal_reasoning":1.1e-07,"input_cache_read":5.5000000000000004e-09,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.00720885,"max_cost":0.0342424775},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":0.00016914178763956407,"request":0.001,"image":4.228544690989102e-05,"web_search":11.839925134769487,"internal_reasoning":0.00016914178763956407,"input_cache_read":8.457089381978204e-06,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":11.084707052958832,"max_cost":52.65303506870501},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.9500000000000006e-08,"completion":3.025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012976128000000002,"max_completion_cost":0.00495616,"max_cost":0.017121280000000003},"sats_pricing":{"prompt":7.611380443780384e-05,"completion":0.00046513991600880125,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.95277715054365,"max_completion_cost":7.6208523838882,"max_cost":26.326580962522872},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1350000000000005e-07,"completion":1.2650000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.041091072000000006,"max_completion_cost":0.12694528000000002,"max_cost":0.13657600000000003},"sats_pricing":{"prompt":0.00048205409477275766,"completion":0.001945130557854987,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":63.18379431005489,"max_completion_cost":195.19774174186367,"max_cost":210.0064435332828},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2","alias_ids":null,"forwarded_model_id":null},{"id":"dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":4.95e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01408,"max_completion_cost":0.00405504,"max_cost":0.01723392},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0007611380443780383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.650148817864203,"max_completion_cost":6.23524285954489,"max_cost":26.49978215306578},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"venice/uncensored","alias_ids":null,"forwarded_model_id":null},{"id":"hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.700000000000001e-08,"completion":3.1350000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.010092544000000002,"max_completion_cost":0.041091072000000006,"max_cost":0.041091072000000006},"sats_pricing":{"prompt":0.00011839925134769487,"completion":0.00048205409477275766,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":15.518826672645062,"max_completion_cost":63.18379431005489,"max_cost":63.18379431005489},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hunyuan-a13b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-large","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.95e-07,"completion":1.0450000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.12976128,"max_completion_cost":0.13697024000000002,"max_cost":0.20185088},"sats_pricing":{"prompt":0.0007611380443780383,"completion":0.0016068469825758589,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":199.52777150543648,"max_completion_cost":210.61264770018298,"max_cost":310.3765334529012},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"morph/morph-v3-large","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-fast","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.4e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0360448,"max_completion_cost":0.02508,"max_cost":0.0444048},"sats_pricing":{"prompt":0.0006765671505582563,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55.424380973732355,"max_completion_cost":38.56432758182061,"max_cost":68.27915683433923},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"morph/morph-v3-fast","alias_ids":null,"forwarded_model_id":null},{"id":"ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":6.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.028413,"max_completion_cost":0.011000000000000001,"max_cost":0.035717000000000006},"sats_pricing":{"prompt":0.00035519775404308456,"completion":0.0010571361727472757,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.689323747299404,"max_completion_cost":16.914178763956407,"max_cost":54.92033844656646},"per_request_limits":null,"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.15625e-08,"completion":1.375e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0132,"max_completion_cost":0.0022528,"max_cost":0.014608},"sats_pricing":{"prompt":7.928521295604566e-05,"completion":0.0002114272345494551,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":20.297014516747687,"max_completion_cost":3.464023810858272,"max_cost":22.462029398534106},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m1","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.025e-07,"completion":1.21e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.3025,"max_completion_cost":0.048400000000000006,"max_cost":0.3388},"sats_pricing":{"prompt":0.00046513991600880125,"completion":0.001860559664035205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":465.1399160088012,"max_completion_cost":74.4223865614082,"max_cost":520.9567059298573},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m1","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.65e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":1.65e-07,"web_search":0.007700000000000001,"internal_reasoning":1.3750000000000002e-06,"input_cache_read":1.65e-08,"input_cache_write":4.583333333333332e-08,"max_prompt_cost":0.17301504,"max_completion_cost":0.09011062500000001,"max_cost":0.25231239},"sats_pricing":{"prompt":0.0002537126814593461,"completion":0.0021142723454945513,"request":0.001,"image":0.0002537126814593461,"web_search":11.839925134769487,"internal_reasoning":0.0021142723454945513,"input_cache_read":2.537126814593461e-05,"input_cache_write":7.047574484981834e-05,"max_prompt_cost":266.0370286739153,"max_completion_cost":138.55883816198542,"max_cost":387.96880625646247},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash:batch","name":"Google: Gemini 2.5 Flash (batch)","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":6.875000000000001e-07,"request":0.0,"image":8.25e-08,"web_search":0.007700000000000001,"internal_reasoning":6.875000000000001e-07,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.08650752,"max_completion_cost":0.04505531250000001,"max_cost":0.126156195},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0010571361727472757,"request":0.001,"image":0.00012685634072967305,"web_search":11.839925134769487,"internal_reasoning":0.0010571361727472757,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0,"max_prompt_cost":133.01851433695765,"max_completion_cost":69.27941908099271,"max_cost":193.98440312823124},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":5.500000000000001e-06,"request":0.0,"image":6.875000000000001e-07,"web_search":0.007700000000000001,"internal_reasoning":5.500000000000001e-06,"input_cache_read":6.875e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.7208960000000001,"max_completion_cost":0.36044800000000005,"max_cost":1.036288},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.008457089381978205,"request":0.001,"image":0.0010571361727472757,"web_search":11.839925134769487,"internal_reasoning":0.008457089381978205,"input_cache_read":0.00010571361727472754,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1108.4876194746473,"max_completion_cost":554.2438097373237,"max_cost":1593.4509529948054},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro:batch","name":"Google: Gemini 2.5 Pro (batch)","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.4375000000000004e-07,"completion":2.7500000000000004e-06,"request":0.0,"image":3.4375000000000004e-07,"web_search":0.007700000000000001,"internal_reasoning":2.7500000000000004e-06,"input_cache_read":6.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.36044800000000005,"max_completion_cost":0.18022400000000002,"max_cost":0.518144},"sats_pricing":{"prompt":0.0005285680863736378,"completion":0.004228544690989103,"request":0.001,"image":0.0005285680863736378,"web_search":11.839925134769487,"internal_reasoning":0.004228544690989103,"input_cache_read":0.00010571361727472754,"input_cache_write":0.0,"max_prompt_cost":554.2438097373237,"max_completion_cost":277.12190486866183,"max_cost":796.7254764974027},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":5.500000000000001e-06,"request":0.0,"image":6.875000000000001e-07,"web_search":0.007700000000000001,"internal_reasoning":5.500000000000001e-06,"input_cache_read":6.875e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.7208960000000001,"max_completion_cost":0.36044800000000005,"max_cost":1.036288},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.008457089381978205,"request":0.001,"image":0.0010571361727472757,"web_search":11.839925134769487,"internal_reasoning":0.008457089381978205,"input_cache_read":0.00010571361727472754,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1108.4876194746473,"max_completion_cost":554.2438097373237,"max_cost":1593.4509529948054},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro-preview-06-05","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.75e-07,"completion":1.1825000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.045056,"max_completion_cost":0.038748160000000004,"max_cost":0.07479296},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.001818274217125314,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002959981283692371,"input_cache_write":0.0,"max_prompt_cost":69.28047621716544,"max_completion_cost":59.58120954676229,"max_cost":115.00559052049465},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1-0528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":8.25e-06,"completion":4.125e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":8.25e-07,"input_cache_write":1.03125e-05,"max_prompt_cost":1.6500000000000001,"max_completion_cost":1.32,"max_cost":2.7060000000000004},"sats_pricing":{"prompt":0.012685634072967307,"completion":0.06342817036483653,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0012685634072967305,"input_cache_write":0.015857042591209132,"max_prompt_cost":2537.1268145934614,"max_completion_cost":2029.701451674769,"max_cost":4160.887975933277},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4-opus-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":2.0625e-06,"max_prompt_cost":0.33,"max_completion_cost":0.528,"max_cost":0.7524000000000001},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":0.0002537126814593461,"input_cache_write":0.0031714085182418267,"max_prompt_cost":507.42536291869226,"max_completion_cost":811.8805806699075,"max_cost":1156.9298274546184},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4-sonnet-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-08,"completion":6.6e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001081344,"max_completion_cost":0.002162688,"max_cost":0.002162688},"sats_pricing":{"prompt":5.074253629186922e-05,"completion":0.00010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.6627314292119706,"max_completion_cost":3.325462858423941,"max_cost":3.325462858423941},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3n-e4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2e-07,"completion":1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.1441792,"max_cost":0.1441792},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.0016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3828357527912815e-05,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":221.69752389492942,"max_cost":221.69752389492942},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":6.875000000000001e-07,"completion":5.500000000000001e-06,"request":0.0,"image":6.875000000000001e-07,"web_search":0.007700000000000001,"internal_reasoning":5.500000000000001e-06,"input_cache_read":6.875e-08,"input_cache_write":2.0625e-07,"max_prompt_cost":0.7208960000000001,"max_completion_cost":0.36044250000000005,"max_cost":1.0362831875},"sats_pricing":{"prompt":0.0010571361727472757,"completion":0.008457089381978205,"request":0.001,"image":0.0010571361727472757,"web_search":11.839925134769487,"internal_reasoning":0.008457089381978205,"input_cache_read":0.00010571361727472754,"input_cache_write":0.00031714085182418263,"max_prompt_cost":1108.4876194746473,"max_completion_cost":554.2353526479417,"max_cost":1593.443553041596},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro-preview-03-25","alias_ids":null,"forwarded_model_id":null},{"id":"virtuoso-large","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":6.6e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0540672,"max_completion_cost":0.04224,"max_cost":0.0699072},"sats_pricing":{"prompt":0.0006342817036483653,"completion":0.0010148507258373844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":83.13657146059853,"max_completion_cost":64.9504464535926,"max_cost":107.49298888069576},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"arcee-ai/virtuoso-large","alias_ids":null,"forwarded_model_id":null},{"id":"llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.900000000000001e-08,"completion":9.900000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01622016,"max_completion_cost":0.0016220160000000002,"max_cost":0.01622016},"sats_pricing":{"prompt":0.00015222760887560768,"completion":0.00015222760887560768,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24.94097143817956,"max_completion_cost":2.494097143817956,"max_cost":24.94097143817956},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-guard-4-12b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":6.6e-08,"completion":2.75e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0027033599999999997,"max_completion_cost":0.0045056,"max_cost":0.006127616000000001},"sats_pricing":{"prompt":0.00010148507258373844,"completion":0.0004228544690989102,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.156828573029927,"max_completion_cost":6.928047621716544,"max_cost":9.422144765534501},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-8b","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":6.435e-08,"completion":2.5025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0084344832,"max_completion_cost":0.002050048,"max_cost":0.009957376},"sats_pricing":{"prompt":9.894794576914499e-05,"completion":0.00038479756688000825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":12.969305147853373,"max_completion_cost":3.1522616678810276,"max_cost":15.310985243993564},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-8b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-14b","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.25125e-07,"completion":5.005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.016400384,"max_completion_cost":0.004100096,"max_cost":0.019475456000000002},"sats_pricing":{"prompt":0.00019239878344000412,"completion":0.0007695951337600165,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25.21809334304822,"max_completion_cost":6.304523335762055,"max_cost":29.946485844869766},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-14b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-32b","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":4.4000000000000004e-08,"completion":1.5400000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00180224,"max_completion_cost":0.0025231360000000005,"max_cost":0.0036044800000000006},"sats_pricing":{"prompt":6.765671505582563e-05,"completion":0.00023679850269538974,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.7712190486866177,"max_completion_cost":3.8797066681612655,"max_cost":5.542438097373236},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-32b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.5025e-07,"completion":1.001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.032800768,"max_completion_cost":0.008200192,"max_cost":0.038950912000000004},"sats_pricing":{"prompt":0.00038479756688000825,"completion":0.001539190267520033,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":50.43618668609644,"max_completion_cost":12.60904667152411,"max_cost":59.89297168973953},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-maverick","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":4.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11534336,"max_completion_cost":0.00720896,"max_cost":0.12075008000000001},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0006765671505582563,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":177.35801911594353,"max_completion_cost":11.084876194746471,"max_cost":185.67167626200342},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-scout","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.65e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0180224,"max_completion_cost":0.00270336,"max_cost":0.019824640000000004},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0002537126814593461,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.712190486866177,"max_completion_cost":4.156828573029927,"max_cost":30.483409535552802},"per_request_limits":null,"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.485e-07,"completion":6.160000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.425e-08,"input_cache_write":0.0,"max_prompt_cost":0.024330240000000003,"max_completion_cost":0.04037017600000001,"max_cost":0.05496832000000001},"sats_pricing":{"prompt":0.0002283414133134115,"completion":0.000947194010781559,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011417070665670575,"input_cache_write":0.0,"max_prompt_cost":37.411457157269346,"max_completion_cost":62.07530669058025,"max_cost":84.52218098494185},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3-0324","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.9305000000000002e-07,"completion":3.0525e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024710400000000004,"max_completion_cost":0.039072,"max_cost":0.039072},"sats_pricing":{"prompt":0.000296843837307435,"completion":0.00046936846069979034,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":37.996011175351676,"max_completion_cost":60.07916296957316,"max_cost":60.07916296957316},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":2.75e-08,"completion":5.5e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.00090112,"max_cost":0.00405504},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":8.457089381978204e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":1.3856095243433089,"max_cost":6.23524285954489},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":2.75e-08,"completion":8.25e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.00135168,"max_cost":0.0045056},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":0.00012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":2.0784142865149633,"max_cost":6.928047621716544},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-12b-it","alias_ids":null,"forwarded_model_id":null},{"id":"command-a","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":5.500000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.35200000000000004,"max_completion_cost":0.045056000000000006,"max_cost":0.385792},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.008457089381978205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":541.253720446605,"max_completion_cost":69.28047621716546,"max_cost":593.2140776094792},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-a-03-2025","alias_ids":null,"forwarded_model_id":null},{"id":"reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":1.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.00720896,"max_cost":0.00720896},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":11.084876194746471,"max_cost":11.084876194746471},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"rekaai/reka-flash-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":4.4000000000000004e-08,"completion":2.475e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.0057671680000000005,"max_completion_cost":0.03244032,"max_cost":0.03244032},"sats_pricing":{"prompt":6.765671505582563e-05,"completion":0.00038056902218901916,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3828357527912815e-05,"input_cache_write":0.0,"max_prompt_cost":8.867900955797177,"max_completion_cost":49.88194287635912,"max_cost":49.88194287635912},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.025e-07,"completion":4.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":0.0,"max_prompt_cost":0.00991232,"max_completion_cost":0.01441792,"max_cost":0.01441792},"sats_pricing":{"prompt":0.00046513991600880125,"completion":0.0006765671505582563,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002114272345494551,"input_cache_write":0.0,"max_prompt_cost":15.2417047677764,"max_completion_cost":22.169752389492942,"max_cost":22.169752389492942},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/skyfall-36b-v2","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":1.1e-06,"completion":4.4e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1408,"max_completion_cost":0.5632,"max_cost":0.5632},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.006765671505582563,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":216.50148817864203,"max_completion_cost":866.0059527145681,"max_cost":866.0059527145681},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-reasoning-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-06,"completion":8.25e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.33,"max_completion_cost":0.066,"max_cost":0.38280000000000003},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.012685634072967307,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":507.42536291869226,"max_completion_cost":101.48507258373844,"max_cost":588.613420985683},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-deep-research","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":1.1e-06,"completion":4.4e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":1.65e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1408,"max_completion_cost":0.5632,"max_cost":0.5632},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.006765671505582563,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.002537126814593461,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":216.50148817864203,"max_completion_cost":866.0059527145681,"max_cost":866.0059527145681},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-deep-research","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-saba","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.01081344,"max_cost":0.01081344},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6914178763956408e-05,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":16.627314292119706,"max_cost":16.627314292119706},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-saba-2502","alias_ids":null,"forwarded_model_id":null},{"id":"aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.4e-07,"completion":8.8e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01441792,"max_completion_cost":0.02883584,"max_cost":0.02883584},"sats_pricing":{"prompt":0.0006765671505582563,"completion":0.0013531343011165126,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.169752389492942,"max_completion_cost":44.339504778985884,"max_cost":44.339504778985884},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":4.125e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0044,"max_completion_cost":0.0132,"max_cost":0.0132},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0006342817036483653,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.765671505582564,"max_completion_cost":20.297014516747687,"max_cost":20.297014516747687},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.4300000000000002e-07,"completion":4.2900000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8600000000000005e-08,"input_cache_write":1.7875e-07,"max_prompt_cost":0.14300000000000002,"max_completion_cost":0.014057472000000001,"max_cost":0.15237164800000003},"sats_pricing":{"prompt":0.00021988432393143332,"completion":0.0006596529717942999,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.397686478628667e-05,"input_cache_write":0.0002748554049142916,"max_prompt_cost":219.88432393143333,"max_completion_cost":21.61550857975562,"max_cost":234.29466298460375},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-01-25","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.75e-08,"completion":4.4000000000000004e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00090112,"max_completion_cost":0.0007208960000000001,"max_cost":0.001171456},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":6.765671505582563e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3856095243433089,"max_completion_cost":1.1084876194746471,"max_cost":1.8012923816463016},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","alias_ids":null,"forwarded_model_id":null},{"id":"sonar","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06988960000000001,"max_completion_cost":0.06988960000000001,"max_cost":0.06988960000000001},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":4.228544690989102,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":107.46592619467344,"max_completion_cost":107.46592619467344,"max_cost":107.46592619467344},"per_request_limits":null,"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":4.4e-07,"completion":4.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.00360448,"max_cost":0.00360448},"sats_pricing":{"prompt":0.0006765671505582563,"completion":0.0006765671505582563,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":5.5424380973732355,"max_cost":5.5424380973732355},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":3.85e-07,"completion":1.3750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024640000000000002,"max_completion_cost":0.022000000000000002,"max_cost":0.04048},"sats_pricing":{"prompt":0.0005919962567384742,"completion":0.0021142723454945513,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":37.88776043126236,"max_completion_cost":33.828357527912814,"max_cost":62.24417785135958},"per_request_limits":null,"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":6.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11002112,"max_completion_cost":0.60511616,"max_cost":0.60511616},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0009302798320176025,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":169.17426286279087,"max_completion_cost":930.4584457453498,"max_cost":930.4584457453498},"per_request_limits":null,"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-01","alias_ids":null,"forwarded_model_id":null},{"id":"phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.850000000000001e-08,"completion":7.700000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0006307840000000001,"max_completion_cost":0.0012615680000000002,"max_cost":0.0012615680000000002},"sats_pricing":{"prompt":5.9199625673847435e-05,"completion":0.00011839925134769487,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.9699266670403164,"max_completion_cost":1.9398533340806328,"max_cost":1.9398533340806328},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"microsoft/phi-4","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.4157000000000002e-07,"completion":5.657850000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5400000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.018120960000000002,"max_completion_cost":0.009052560000000001,"max_cost":0.024908400000000004},"sats_pricing":{"prompt":0.000217685480692119,"completion":0.0008699807847240979,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3679850269538972e-05,"input_cache_write":0.0,"max_prompt_cost":27.86374152859123,"max_completion_cost":13.919692555585566,"max_cost":38.300466393102894},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3","alias_ids":null,"forwarded_model_id":null},{"id":"l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":3.575e-07,"completion":4.125e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04685824,"max_completion_cost":0.0067584,"max_cost":0.04775936},"sats_pricing":{"prompt":0.0005497108098285832,"completion":0.0006342817036483653,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.05169526585206,"max_completion_cost":10.392071432574816,"max_cost":73.43730479019537},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","alias_ids":null,"forwarded_model_id":null},{"id":"command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":2.0625e-08,"completion":8.25e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00264,"max_completion_cost":0.00033,"max_cost":0.0028875000000000003},"sats_pricing":{"prompt":3.171408518241826e-05,"completion":0.00012685634072967305,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.059402903349538,"max_completion_cost":0.5074253629186922,"max_cost":4.439971925538558},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r7b-12-2024","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.5e-08,"completion":1.7600000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00720896,"max_completion_cost":0.0028835840000000002,"max_cost":0.009191424},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0002706268602233025,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.084876194746471,"max_completion_cost":4.4339504778985885,"max_cost":14.133217148301751},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.3-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"nova-lite-v1","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.3e-08,"completion":1.32e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.009899999999999999,"max_completion_cost":0.0006758399999999999,"max_cost":0.01040688},"sats_pricing":{"prompt":5.074253629186922e-05,"completion":0.00020297014516747688,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":15.222760887560765,"max_completion_cost":1.0392071432574816,"max_cost":16.002166245003878},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-micro-v1","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":1.9250000000000004e-08,"completion":7.700000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0024640000000000005,"max_completion_cost":0.0003942400000000001,"max_cost":0.0027596800000000005},"sats_pricing":{"prompt":2.9599812836923718e-05,"completion":0.00011839925134769487,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.788776043126236,"max_completion_cost":0.6062041669001977,"max_cost":4.243429168301384},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-micro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-pro-v1","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":4.4e-07,"completion":1.76e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.132,"max_completion_cost":0.0090112,"max_cost":0.1387584},"sats_pricing":{"prompt":0.0006765671505582563,"completion":0.002706268602233025,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":202.9701451674769,"max_completion_cost":13.856095243433089,"max_cost":213.3622166000517},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-pro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2407","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.1441792,"max_completion_cost":0.4325376,"max_cost":0.4325376},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":221.69752389492942,"max_completion_cost":665.0925716847883,"max_cost":665.0925716847883},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large-2407","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.6300000000000006e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.011894784000000002,"max_completion_cost":0.0180224,"max_cost":0.0180224},"sats_pricing":{"prompt":0.0005581678992105615,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.29004572133168,"max_completion_cost":27.712190486866177,"max_cost":27.712190486866177},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.2e-07,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.22528,"max_completion_cost":0.22528,"max_cost":0.22528},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":346.40238108582724,"max_completion_cost":346.40238108582724,"max_cost":346.40238108582724},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":1024000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/unslopnemo-12b","alias_ids":null,"forwarded_model_id":null},{"id":"magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.65e-06,"completion":2.7500000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0270336,"max_completion_cost":0.005632000000000001,"max_cost":0.029286400000000004},"sats_pricing":{"prompt":0.002537126814593461,"completion":0.004228544690989103,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":41.568285730299266,"max_completion_cost":8.660059527145682,"max_cost":45.032309541157545},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthracite-org/magnum-v4-72b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":5.5e-08,"completion":1.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00180224,"max_completion_cost":0.00360448,"max_cost":0.00360448},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.00016914178763956407,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.7712190486866177,"max_completion_cost":5.5424380973732355,"max_cost":5.5424380973732355},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-7b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"rocinante-12b","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.375e-07,"completion":2.75e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0090112,"max_completion_cost":0.0180224,"max_cost":0.0180224},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0004228544690989102,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.856095243433089,"max_completion_cost":27.712190486866177,"max_cost":27.712190486866177},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/rocinante-12b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":1.485e-08,"completion":1.1055000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0008910000000000001,"max_completion_cost":0.006633000000000001,"max_cost":0.006633000000000001},"sats_pricing":{"prompt":2.283414133134115e-05,"completion":0.00016998749657776193,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.370048479880469,"max_completion_cost":10.199249794665715,"max_cost":10.199249794665715},"per_request_limits":null,"top_provider":{"context_length":60000,"max_completion_tokens":60000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.2-1b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.75e-08,"completion":1.8150000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.023789568000000004,"max_cost":0.023789568000000004},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":0.00027908394960528076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":36.58009144266336,"max_cost":36.58009144266336},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.2-3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.9800000000000003e-07,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.006488064000000001,"max_completion_cost":0.00360448,"max_cost":0.006848512000000001},"sats_pricing":{"prompt":0.00030445521775121536,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.976388575271825,"max_completion_cost":5.5424380973732355,"max_cost":10.530632385009149},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-08-2024","name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":3.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01056,"max_completion_cost":0.00132,"max_cost":0.011550000000000001},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.23761161339815,"max_completion_cost":2.029701451674769,"max_cost":17.75988770215423},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":1.3750000000000002e-06,"completion":5.500000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.17600000000000002,"max_completion_cost":0.022000000000000002,"max_cost":0.1925},"sats_pricing":{"prompt":0.0021142723454945513,"completion":0.008457089381978205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":270.6268602233025,"max_completion_cost":33.828357527912814,"max_cost":295.9981283692371},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r-plus-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.675e-07,"completion":4.675e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06127616,"max_completion_cost":0.00765952,"max_cost":0.06127616},"sats_pricing":{"prompt":0.0007188525974681473,"completion":0.0007188525974681473,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":94.221447655345,"max_completion_cost":11.777680956918125,"max_cost":94.221447655345},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3.1-euryale-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":3.85e-07,"completion":3.85e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05046272,"max_completion_cost":0.00630784,"max_cost":0.05046272},"sats_pricing":{"prompt":0.0005919962567384742,"completion":0.0005919962567384742,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":77.5941333632253,"max_completion_cost":9.699266670403162,"max_cost":77.5941333632253},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":5.5e-07,"completion":5.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0720896,"max_completion_cost":0.0090112,"max_cost":0.0720896},"sats_pricing":{"prompt":0.0008457089381978204,"completion":0.0008457089381978204,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":110.84876194746471,"max_completion_cost":13.856095243433089,"max_cost":110.84876194746471},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","alias_ids":null,"forwarded_model_id":null},{"id":"l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.2000000000000002e-08,"completion":2.75e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00018022400000000001,"max_completion_cost":0.00022528,"max_cost":0.00022528},"sats_pricing":{"prompt":3.3828357527912815e-05,"completion":4.228544690989102e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2771219048686618,"max_completion_cost":0.3464023810858272,"max_cost":0.3464023810858272},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3-lunaris-8b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.2e-07,"completion":2.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02883584,"max_completion_cost":0.00360448,"max_cost":0.02883584},"sats_pricing":{"prompt":0.00033828357527912814,"completion":0.00033828357527912814,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":44.339504778985884,"max_completion_cost":5.5424380973732355,"max_cost":44.339504778985884},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.1-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.75e-08,"completion":4.4000000000000004e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.00360448,"max_completion_cost":0.0057671680000000005,"max_cost":0.0057671680000000005},"sats_pricing":{"prompt":4.228544690989102e-05,"completion":6.765671505582563e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.114272345494551e-05,"input_cache_write":0.0,"max_prompt_cost":5.5424380973732355,"max_completion_cost":8.867900955797177,"max_cost":8.867900955797177},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.1-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-nemo","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":1.0450000000000001e-08,"completion":1.65e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0013697024000000001,"max_completion_cost":0.000270336,"max_cost":0.0014688256},"sats_pricing":{"prompt":1.606846982575859e-05,"completion":2.537126814593461e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.1061264770018298,"max_completion_cost":0.41568285730299265,"max_cost":2.2585435246795935},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-nemo","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":3.575e-07,"completion":3.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00292864,"max_completion_cost":0.00073216,"max_cost":0.00292864},"sats_pricing":{"prompt":0.0005497108098285832,"completion":0.0005497108098285832,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.503230954115754,"max_completion_cost":1.1258077385289385,"max_cost":4.503230954115754},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-2-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.0720896,"max_completion_cost":0.2162688,"max_cost":0.2162688},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":110.84876194746471,"max_completion_cost":332.5462858423941,"max_cost":332.5462858423941},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mixtral-8x22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"wizardlm-2-8x22b","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":3.4100000000000005e-07,"completion":3.4100000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.022347435000000002,"max_completion_cost":0.0027280000000000004,"max_cost":0.022347435000000006},"sats_pricing":{"prompt":0.0005243395416826487,"completion":0.0005243395416826487,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.36259186417238,"max_completion_cost":4.19471633346119,"max_cost":34.36259186417239},"per_request_limits":null,"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"microsoft/wizardlm-2-8x22b","alias_ids":null,"forwarded_model_id":null},{"id":"claude-3-haiku","name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.375e-07,"completion":6.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":1.65e-07,"max_prompt_cost":0.0275,"max_completion_cost":0.0028160000000000004,"max_cost":0.0297528},"sats_pricing":{"prompt":0.0002114272345494551,"completion":0.0010571361727472757,"request":0.001,"image":0.0,"web_search":8.457089381978204,"internal_reasoning":0.0,"input_cache_read":2.537126814593461e-05,"input_cache_write":0.0002537126814593461,"max_prompt_cost":42.28544690989102,"max_completion_cost":4.330029763572841,"max_cost":45.74947072074929},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-3-haiku","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large","name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1e-06,"completion":3.3e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.1408,"max_completion_cost":0.4224,"max_cost":0.4224},"sats_pricing":{"prompt":0.0016914178763956407,"completion":0.005074253629186922,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016914178763956407,"input_cache_write":0.0,"max_prompt_cost":216.50148817864203,"max_completion_cost":649.504464535926,"max_cost":649.504464535926},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large","alias_ids":null,"forwarded_model_id":null},{"id":"weaver","name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":2.75e-07,"completion":4.125e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0022,"max_completion_cost":0.002475,"max_cost":0.0030250000000000003},"sats_pricing":{"prompt":0.0004228544690989102,"completion":0.0006342817036483653,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.382835752791282,"max_completion_cost":3.805690221890192,"max_cost":4.651399160088013},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":6000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mancer/weaver","alias_ids":null,"forwarded_model_id":null},{"id":"remm-slerp-l2-13b","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":2.475e-07,"completion":3.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0015206400000000002,"max_completion_cost":0.0021964800000000002,"max_cost":0.0021964800000000002},"sats_pricing":{"prompt":0.00038056902218901916,"completion":0.0005497108098285832,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.338216072329334,"max_completion_cost":3.377423215586816,"max_cost":3.377423215586816},"per_request_limits":null,"top_provider":{"context_length":6144,"max_completion_tokens":6144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"undi95/remm-slerp-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"mythomax-l2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":4.4000000000000004e-08,"completion":6.05e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00018022400000000001,"max_completion_cost":0.000247808,"max_cost":0.000247808},"sats_pricing":{"prompt":6.765671505582563e-05,"completion":9.302798320176024e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2771219048686618,"max_completion_cost":0.38104261919440996,"max_cost":0.38104261919440996},"per_request_limits":null,"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"gryphe/mythomax-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-multimodal-3.5","name":"VoyageAI by MongoDB: voyage-multimodal-3.5","created":1785188629,"description":"voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...","context_length":32000,"architecture":{"modality":"text+image->embeddings","input_modalities":["text","image"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.6e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0021119999999999997,"max_completion_cost":0.0,"max_cost":0.0021119999999999997},"sats_pricing":{"prompt":0.00010148507258373844,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.2475223226796297,"max_completion_cost":0.0,"max_cost":3.2475223226796297},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-multimodal-3.5-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-lite","name":"VoyageAI by MongoDB: voyage-4-lite","created":1785188627,"description":"voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1000000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00035200000000000005,"max_completion_cost":0.0,"max_cost":0.00035200000000000005},"sats_pricing":{"prompt":1.6914178763956408e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5412537204466051,"max_completion_cost":0.0,"max_cost":0.5412537204466051},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-lite-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4","name":"VoyageAI by MongoDB: voyage-4","created":1785188626,"description":"voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.3e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0010559999999999999,"max_completion_cost":0.0,"max_cost":0.0010559999999999999},"sats_pricing":{"prompt":5.074253629186922e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.6237611613398149,"max_completion_cost":0.0,"max_cost":1.6237611613398149},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-large","name":"VoyageAI by MongoDB: voyage-4-large","created":1785188624,"description":"voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.6e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0021119999999999997,"max_completion_cost":0.0,"max_cost":0.0021119999999999997},"sats_pricing":{"prompt":0.00010148507258373844,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.2475223226796297,"max_completion_cost":0.0,"max_cost":3.2475223226796297},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-large-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2","name":"Google: Gemini Embedding 2","created":1779290135,"description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":0.0,"request":0.0,"image":2.475e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00090112,"max_completion_cost":0.0,"max_cost":0.00090112},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0,"request":0.001,"image":0.00038056902218901916,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3856095243433089,"max_completion_cost":0.0,"max_cost":1.3856095243433089},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2-preview","name":"Google: Gemini Embedding 2 Preview","created":1776436465,"description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1e-07,"completion":0.0,"request":0.0,"image":2.475e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00090112,"max_completion_cost":0.0,"max_cost":0.00090112},"sats_pricing":{"prompt":0.00016914178763956407,"completion":0.0,"request":0.001,"image":0.00038056902218901916,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3856095243433089,"max_completion_cost":0.0,"max_cost":1.3856095243433089},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2-preview","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-4b","name":"Perplexity: Embed V1 4B","created":1773625372,"description":"pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.65e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0005279999999999999,"max_completion_cost":0.0,"max_cost":0.0005279999999999999},"sats_pricing":{"prompt":2.537126814593461e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.8118805806699074,"max_completion_cost":0.0,"max_cost":0.8118805806699074},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-4B","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-0.6b","name":"Perplexity: Embed V1 0.6B","created":1773624868,"description":"pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2000000000000003e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.04e-05,"max_completion_cost":0.0,"max_cost":7.04e-05},"sats_pricing":{"prompt":3.382835752791282e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.108250744089321,"max_completion_cost":0.0,"max_cost":0.108250744089321},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-0.6B","alias_ids":null,"forwarded_model_id":null},{"id":"gte-base","name":"Thenlper: GTE-Base","created":1763433820,"description":"The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-base-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"gte-large","name":"Thenlper: GTE-Large","created":1763433655,"description":"The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8160000000000002e-06,"max_completion_cost":0.0,"max_cost":2.8160000000000002e-06},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00433002976357284,"max_completion_cost":0.0,"max_cost":0.00433002976357284},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-large-v2","name":"Intfloat: E5-Large-v2","created":1763433432,"description":"The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8160000000000002e-06,"max_completion_cost":0.0,"max_cost":2.8160000000000002e-06},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00433002976357284,"max_completion_cost":0.0,"max_cost":0.00433002976357284},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-large-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-base-v2","name":"Intfloat: E5-Base-v2","created":1763433192,"description":"The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multilingual-e5-large","name":"Intfloat: Multilingual-E5-Large","created":1763433047,"description":"The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8160000000000002e-06,"max_completion_cost":0.0,"max_cost":2.8160000000000002e-06},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00433002976357284,"max_completion_cost":0.0,"max_cost":0.00433002976357284},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/multilingual-e5-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"paraphrase-minilm-l6-v2","name":"Sentence Transformers: paraphrase-MiniLM-L6-v2","created":1763432454,"description":"The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/paraphrase-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l12-v2","name":"Sentence Transformers: all-MiniLM-L12-v2","created":1763432155,"description":"The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l12-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-base-en-v1.5","name":"BAAI: bge-base-en-v1.5","created":1763431837,"description":"The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-base-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multi-qa-mpnet-base-dot-v1","name":"Sentence Transformers: multi-qa-mpnet-base-dot-v1","created":1763431339,"description":"The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-large-en-v1.5","name":"BAAI: bge-large-en-v1.5","created":1763431087,"description":"The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8160000000000002e-06,"max_completion_cost":0.0,"max_cost":2.8160000000000002e-06},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00433002976357284,"max_completion_cost":0.0,"max_cost":0.00433002976357284},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-large-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-m3","name":"BAAI: bge-m3","created":1763424372,"description":"The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.","context_length":8194,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.5056000000000004e-05,"max_completion_cost":0.0,"max_cost":4.5056000000000004e-05},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06928047621716545,"max_completion_cost":0.0,"max_cost":0.06928047621716545},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-m3-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-mpnet-base-v2","name":"Sentence Transformers: all-mpnet-base-v2","created":1763421830,"description":"The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-mpnet-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l6-v2","name":"Sentence Transformers: all-MiniLM-L6-v2","created":1763421176,"description":"The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7500000000000002e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4080000000000001e-06,"max_completion_cost":0.0,"max_cost":1.4080000000000001e-06},"sats_pricing":{"prompt":4.228544690989102e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00216501488178642,"max_completion_cost":0.0,"max_cost":0.00216501488178642},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-embed-2312","name":"Mistral: Mistral Embed 2312","created":1761944622,"description":"Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.5e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00045056,"max_completion_cost":0.0,"max_cost":0.00045056},"sats_pricing":{"prompt":8.457089381978204e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6928047621716544,"max_completion_cost":0.0,"max_cost":0.6928047621716544},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-embed-2312","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-001","name":"Google: Gemini Embedding 001","created":1761943410,"description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","context_length":20000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00165,"max_completion_cost":0.0,"max_cost":0.00165},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.537126814593461,"max_completion_cost":0.0,"max_cost":2.537126814593461},"per_request_limits":null,"top_provider":{"context_length":20000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-001","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-embed-2505","name":"Mistral: Codestral Embed 2505","created":1761864460,"description":"Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":8.25e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00067584,"max_completion_cost":0.0,"max_cost":0.00067584},"sats_pricing":{"prompt":0.00012685634072967305,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.0392071432574816,"max_completion_cost":0.0,"max_cost":1.0392071432574816},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/codestral-embed-2505","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-8b","name":"Qwen: Qwen3 Embedding 8B","created":1761680622,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.5000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00017600000000000002,"max_completion_cost":0.0,"max_cost":0.00017600000000000002},"sats_pricing":{"prompt":8.457089381978204e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27062686022330257,"max_completion_cost":0.0,"max_cost":0.27062686022330257},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-4b","name":"Qwen: Qwen3 Embedding 4B","created":1761662922,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1000000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00036044800000000003,"max_completion_cost":0.0,"max_cost":0.00036044800000000003},"sats_pricing":{"prompt":1.6914178763956408e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5542438097373236,"max_completion_cost":0.0,"max_cost":0.5542438097373236},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-4b","alias_ids":null,"forwarded_model_id":null}]}