{"data":[{"id":"muse-spark-1.2","name":"Meta: Muse Spark 1.2","created":1785959287,"description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":4.887499999999999e-06,"request":0.0,"image":0.0,"web_search":0.002875,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":1.507328,"max_completion_cost":5.124915199999999,"max_cost":5.124915199999999},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.007520221703059795,"request":0.001,"image":0.0,"web_search":4.423659825329292,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":2319.2717625022437,"max_completion_cost":7885.523992507628,"max_cost":7885.523992507628},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta/muse-spark-1.2-20260805","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.8-max","name":"Qwen: Qwen3.8 Max","created":1785731612,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":2.875e-06,"max_prompt_cost":2.2999999999999994,"max_completion_cost":0.9043968,"max_cost":2.9029311999999994},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.004423659825329292,"max_prompt_cost":3538.9278602634326,"max_completion_cost":1391.5630575013463,"max_cost":4466.63656526433},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.8-max-20260803","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","created":1785606009,"description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0303999999999999e-07,"completion":2.0607999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0607999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.10804527103999999,"max_completion_cost":0.027011317759999997,"max_cost":0.12155092992},"sats_pricing":{"prompt":0.0001585439681398018,"completion":0.0003170879362796036,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.170879362796036e-05,"input_cache_write":0.0,"max_prompt_cost":166.24539993616082,"max_completion_cost":41.561349984040206,"max_cost":187.02607492818095},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~deepseek/deepseek-v4-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.035e-07,"completion":2.07e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0699999999999997e-08,"input_cache_write":0.0,"max_prompt_cost":0.108527616,"max_completion_cost":0.07948799999999999,"max_cost":0.148271616},"sats_pricing":{"prompt":0.0001592517537118545,"completion":0.000318503507423709,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.18503507423709e-05,"input_cache_write":0.0,"max_prompt_cost":166.98756690016154,"max_completion_cost":122.30534685070425,"max_cost":228.14024032551367},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v4-flash-20260731","alias_ids":null,"forwarded_model_id":null},{"id":"inkling-small","name":"Thinking Machines: Inkling Small","created":1785443117,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.174999999999999e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.27131903999999996,"max_completion_cost":0.36175872,"max_cost":0.49741823999999996},"sats_pricing":{"prompt":0.0007962587685592724,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":417.4689172504038,"max_completion_cost":556.6252230005385,"max_cost":765.3596816257404},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thinkingmachines/inkling-small-20260730","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-flash","name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.449999999999999e-08,"completion":1.4949999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.9e-09,"input_cache_write":4.37e-08,"max_prompt_cost":0.03449999999999999,"max_completion_cost":0.009797631999999999,"max_cost":0.04203663999999999},"sats_pricing":{"prompt":5.308391790395149e-05,"completion":0.00023003031091712315,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.06167835807903e-05,"input_cache_write":6.723962934500524e-05,"max_prompt_cost":53.08391790395149,"max_completion_cost":15.075266456264583,"max_cost":64.68027671646271},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.7-flash-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5-fast","name":"Claude Opus 5 (Fast)","created":1784912546,"description":"Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":5.7499999999999995e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-06,"input_cache_write":1.4374999999999999e-05,"max_prompt_cost":11.5,"max_completion_cost":7.359999999999999,"max_cost":17.387999999999998},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.08847319650658583,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0017694639301317164,"input_cache_write":0.022118299126646458,"max_prompt_cost":17694.63930131717,"max_completion_cost":11324.569152842987,"max_cost":26754.294623591555},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-opus-5-fast-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5","name":"Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5:batch","name":"Claude Opus 5 (batch)","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":3.5937499999999997e-06,"max_prompt_cost":2.875,"max_completion_cost":1.8399999999999999,"max_cost":4.3469999999999995},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0055295747816616145,"max_prompt_cost":4423.659825329292,"max_completion_cost":2831.1422882107468,"max_cost":6688.573655897889},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"ling-3.0-flash","name":"Ling-3.0-flash","created":1784818580,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.4149999999999998e-08,"completion":7.244999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.83e-09,"input_cache_write":0.0,"max_prompt_cost":0.0063307775999999994,"max_completion_cost":0.0023740415999999997,"max_cost":0.007913472},"sats_pricing":{"prompt":3.715874253276605e-05,"completion":0.00011147622759829815,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.431748506553211e-06,"input_cache_write":0.0,"max_prompt_cost":9.740941402509423,"max_completion_cost":3.6528530259410337,"max_cost":12.17617675313678},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"inclusionai/ling-3.0-flash-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-s-2.1","name":"Poolside: Laguna S 2.1","created":1784652683,"description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.035e-07,"completion":2.07e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0349999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.108527616,"max_completion_cost":0.027131904,"max_cost":0.122093568},"sats_pricing":{"prompt":0.0001592517537118545,"completion":0.000318503507423709,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.592517537118545e-05,"input_cache_write":0.0,"max_prompt_cost":166.98756690016154,"max_completion_cost":41.746891725040385,"max_cost":187.86101276268175},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"poolside/laguna-s-2.1-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash","name":"Google: Gemini 3.6 Flash","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":8.625e-06,"request":0.0,"image":1.725e-06,"web_search":0.0161,"internal_reasoning":8.625e-06,"input_cache_read":1.725e-07,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":1.8087936,"max_completion_cost":0.565248,"max_cost":2.260992},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.013270979475987875,"request":0.001,"image":0.002654195895197575,"web_search":24.772495021844033,"internal_reasoning":0.013270979475987875,"input_cache_read":0.00026541958951975753,"input_cache_write":0.00014745532751097632,"max_prompt_cost":2783.1261150026926,"max_completion_cost":869.7269109383413,"max_cost":3478.9076437533654},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash:batch","name":"Google: Gemini 3.6 Flash (batch)","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.625e-07,"completion":4.3125e-06,"request":0.0,"image":8.625e-07,"web_search":0.0161,"internal_reasoning":4.3125e-06,"input_cache_read":8.625e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.9043968,"max_completion_cost":0.282624,"max_cost":1.130496},"sats_pricing":{"prompt":0.0013270979475987876,"completion":0.006635489737993937,"request":0.001,"image":0.0013270979475987876,"web_search":24.772495021844033,"internal_reasoning":0.006635489737993937,"input_cache_read":0.00013270979475987876,"input_cache_write":0.00014745532751097632,"max_prompt_cost":1391.5630575013463,"max_completion_cost":434.8634554691707,"max_cost":1739.4538218766827},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite","name":"Google: Gemini 3.5 Flash Lite","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.875e-06,"request":0.0,"image":3.45e-07,"web_search":0.0161,"internal_reasoning":2.875e-06,"input_cache_read":3.449999999999999e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.36175872,"max_completion_cost":0.188416,"max_cost":0.5275648},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.004423659825329292,"request":0.001,"image":0.0005308391790395151,"web_search":24.772495021844033,"internal_reasoning":0.004423659825329292,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":556.6252230005385,"max_completion_cost":289.90897031278047,"max_cost":811.7451168757852},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite:batch","name":"Google: Gemini 3.5 Flash Lite (batch)","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.4375e-06,"request":0.0,"image":1.725e-07,"web_search":0.0161,"internal_reasoning":1.4375e-06,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":0.094208,"max_cost":0.2637824},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.002211829912664646,"request":0.001,"image":0.00026541958951975753,"web_search":24.772495021844033,"internal_reasoning":0.002211829912664646,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":144.95448515639023,"max_cost":405.8725584378926},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"longcat-2.0","name":"Meituan: LongCat 2.0","created":1784554658,"description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.9e-09,"input_cache_write":0.0,"max_prompt_cost":0.36182082,"max_completion_cost":0.36175872,"max_cost":0.63313986},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.06167835807903e-05,"input_cache_write":0.0,"max_prompt_cost":556.7207740527656,"max_completion_cost":556.6252230005385,"max_cost":974.1896913031695},"per_request_limits":null,"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meituan/longcat-2.0-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"inkling","name":"Thinking Machines: Inkling","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0925e-06,"completion":4.6575e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.8399999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.57278464,"max_completion_cost":1.22093568,"max_cost":1.507328},"sats_pricing":{"prompt":0.0016809907336251307,"completion":0.007166328917033453,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00028311422882107466,"input_cache_write":0.0,"max_prompt_cost":881.3232697508525,"max_completion_cost":1878.6101276268175,"max_cost":2319.2717625022437},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"inkling:batch","name":"Thinking Machines: Inkling (batch)","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":2.32875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.774999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.30146559999999994,"max_completion_cost":1.22093568,"max_cost":1.22093568},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.0035831644585167266,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001504044340611959,"input_cache_write":0.0,"max_prompt_cost":463.85435250044867,"max_completion_cost":1878.6101276268175,"max_cost":1878.6101276268175},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"auto-beta","name":"Auto Router (Beta)","created":1784311165,"description":"Auto Router (Beta) is a task-aware router from OpenRouter. It classifies each request, then routes it the [most popular model](/rankings#task-spend) for that task based on aggregate spend, filtered by your...","context_length":2000000,"architecture":{"modality":"text+image+file+audio+video->text+image","input_modalities":["text","image","audio","file","video"],"output_modalities":["text","image"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":-1.15,"completion":-1.15,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-2300000.0,"max_completion_cost":-2300000.0,"max_cost":-2300000.0},"sats_pricing":{"prompt":-1769.4639301317166,"completion":-1769.4639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-3538927860.2634335,"max_completion_cost":-3538927860.2634335,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/auto-beta","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k3","name":"MoonshotAI: Kimi K3","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":0.0,"max_prompt_cost":3.6175872,"max_completion_cost":18.087936,"max_cost":18.087936},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.0,"max_prompt_cost":5566.252230005385,"max_completion_cost":27831.261150026923,"max_cost":27831.261150026923},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k3-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.1","name":"Meta: Muse Spark 1.1","created":1784215741,"description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":4.887499999999999e-06,"request":0.0,"image":0.0,"web_search":0.002875,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":1.507328,"max_completion_cost":5.124915199999999,"max_cost":5.124915199999999},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.007520221703059795,"request":0.001,"image":0.0,"web_search":4.423659825329292,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":2319.2717625022437,"max_completion_cost":7885.523992507628,"max_cost":7885.523992507628},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta/muse-spark-1.1-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-air-v2.5","name":"Kwaipilot: KAT-Coder-Air V2.5","created":1783714590,"description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.04416,"max_completion_cost":0.0552,"max_cost":0.08556},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":67.94741491705793,"max_completion_cost":84.93426864632241,"max_cost":131.6481164017997},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"kwaipilot/kat-coder-air-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","created":1783714589,"description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.51e-07,"completion":3.404e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":0.217856,"max_completion_cost":0.27232,"max_cost":0.422096},"sats_pricing":{"prompt":0.0013094033082974705,"completion":0.005237613233189882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":335.2072469241524,"max_completion_cost":419.00905865519053,"max_cost":649.4640409155454},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"kwaipilot/kat-coder-pro-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":1.4374999999999997e-07,"max_prompt_cost":0.12074999999999998,"max_completion_cost":0.08832,"max_cost":0.19434999999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.00022118299126646455,"max_prompt_cost":185.79371266383023,"max_completion_cost":135.89482983411585,"max_cost":299.03940419226006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro:batch","name":"OpenAI: GPT-5.6 Luna Pro (batch)","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.12074999999999998,"max_completion_cost":0.08832,"max_cost":0.19434999999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":185.79371266383023,"max_completion_cost":135.89482983411585,"max_cost":299.03940419226006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna","name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":1.4374999999999997e-07,"max_prompt_cost":0.12074999999999998,"max_completion_cost":0.08832,"max_cost":0.19434999999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.00022118299126646455,"max_prompt_cost":185.79371266383023,"max_completion_cost":135.89482983411585,"max_cost":299.03940419226006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna:batch","name":"OpenAI: GPT-5.6 Luna (batch)","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.12074999999999998,"max_completion_cost":0.08832,"max_cost":0.19434999999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":185.79371266383023,"max_completion_cost":135.89482983411585,"max_cost":299.03940419226006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":1.4375e-06,"max_prompt_cost":1.2074999999999998,"max_completion_cost":0.8832,"max_cost":1.9434999999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.002211829912664646,"max_prompt_cost":1857.9371266383023,"max_completion_cost":1358.9482983411585,"max_cost":2990.394041922601},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro:batch","name":"OpenAI: GPT-5.6 Terra Pro (batch)","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":1.2074999999999998,"max_completion_cost":0.8832,"max_cost":1.9434999999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":1857.9371266383023,"max_completion_cost":1358.9482983411585,"max_cost":2990.394041922601},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra","name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":1.4375e-06,"max_prompt_cost":1.2074999999999998,"max_completion_cost":0.8832,"max_cost":1.9434999999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.002211829912664646,"max_prompt_cost":1857.9371266383023,"max_completion_cost":1358.9482983411585,"max_cost":2990.394041922601},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra:batch","name":"OpenAI: GPT-5.6 Terra (batch)","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":1.2074999999999998,"max_completion_cost":0.8832,"max_cost":1.9434999999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":1857.9371266383023,"max_completion_cost":1358.9482983411585,"max_cost":2990.394041922601},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":6.0375,"max_completion_cost":4.4159999999999995,"max_cost":9.7175},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":9289.685633191513,"max_completion_cost":6794.741491705791,"max_cost":14951.970209613006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro:batch","name":"OpenAI: GPT-5.6 Sol Pro (batch)","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":3.01875,"max_completion_cost":2.2079999999999997,"max_cost":4.85875},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":4644.842816595757,"max_completion_cost":3397.3707458528957,"max_cost":7475.985104806503},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol","name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":6.0375,"max_completion_cost":4.4159999999999995,"max_cost":9.7175},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":9289.685633191513,"max_completion_cost":6794.741491705791,"max_cost":14951.970209613006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol:batch","name":"OpenAI: GPT-5.6 Sol (batch)","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":3.01875,"max_completion_cost":2.2079999999999997,"max_cost":4.85875},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":4644.842816595757,"max_completion_cost":3397.3707458528957,"max_cost":7475.985104806503},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.5","name":"SpaceXAI: Grok 4.5","created":1783523154,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":0.0,"max_prompt_cost":1.1499999999999997,"max_completion_cost":3.45,"max_cost":3.45},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.0,"max_prompt_cost":1769.4639301317163,"max_completion_cost":5308.391790395151,"max_cost":5308.391790395151},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"x-ai/grok-4.5-20260708","alias_ids":null,"forwarded_model_id":null},{"id":"grok-latest","name":"xAI: Grok Latest","created":1783519360,"description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":0.0,"max_prompt_cost":1.1499999999999997,"max_completion_cost":3.45,"max_cost":3.45},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.0,"max_prompt_cost":1769.4639301317163,"max_completion_cost":5308.391790395151,"max_cost":5308.391790395151},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~x-ai/grok-latest","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","created":1783443096,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.049999999999999e-07,"completion":1.6099999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.07e-07,"input_cache_write":0.0,"max_prompt_cost":0.10551295999999999,"max_completion_cost":0.052756479999999994,"max_cost":0.1318912},"sats_pricing":{"prompt":0.0012386247510922017,"completion":0.0024772495021844034,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000318503507423709,"input_cache_write":0.0,"max_prompt_cost":162.34902337515706,"max_completion_cost":81.17451168757853,"max_cost":202.9362792189463},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"aion-labs/aion-3.0-mini-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0","name":"AionLabs: Aion-3.0","created":1783443095,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.4521984,"max_completion_cost":0.2260992,"max_cost":0.565248},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0013270979475987876,"input_cache_write":0.0,"max_prompt_cost":695.7815287506731,"max_completion_cost":347.8907643753366,"max_cost":869.7269109383413},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"aion-labs/aion-3.0-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"hy3","name":"Tencent: Hy3","created":1783344048,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.5179999999999997e-07,"completion":6.071999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.794999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.03979345919999999,"max_completion_cost":0.07772159999999999,"max_cost":0.09808465919999998},"sats_pricing":{"prompt":0.00023356923877738658,"completion":0.0009342769551095463,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.8392309694346644e-05,"input_cache_write":0.0,"max_prompt_cost":61.228774530059226,"max_completion_cost":119.58745025402193,"max_cost":150.91936222057566},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"tencent/hy3-20260706","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","created":1783002429,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.899999999999998e-08,"completion":1.3799999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.018087935999999995,"max_completion_cost":0.004521983999999999,"max_cost":0.020348927999999995},"sats_pricing":{"prompt":0.00010616783580790298,"completion":0.00021233567161580596,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":27.83126115002692,"max_completion_cost":6.95781528750673,"max_cost":31.310168793780285},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"poolside/laguna-xs-2.1-20260625","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":2.875e-06,"max_prompt_cost":2.2999999999999994,"max_completion_cost":1.472,"max_cost":3.4776},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.004423659825329292,"max_prompt_cost":3538.9278602634326,"max_completion_cost":2264.9138305685974,"max_cost":5350.858924718311},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5:batch","name":"Anthropic: Claude Sonnet 5 (batch)","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":1.4375e-06,"max_prompt_cost":1.1499999999999997,"max_completion_cost":0.736,"max_cost":1.7388},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.002211829912664646,"max_prompt_cost":1769.4639301317163,"max_completion_cost":1132.4569152842987,"max_cost":2675.4294623591554},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","created":1782837225,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.725e-06,"request":0.0,"image":0.0,"web_search":0.0161,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018841599999999997,"max_completion_cost":0.1130496,"max_cost":0.1130496},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.002654195895197575,"request":0.001,"image":0.0,"web_search":24.772495021844033,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.990897031278042,"max_completion_cost":173.9453821876683,"max_cost":173.9453821876683},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","created":1782312964,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.8749999999999996e-08,"completion":1.1499999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999997e-09,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.030146559999999996,"max_cost":0.030146559999999996},"sats_pricing":{"prompt":4.4236598253292915e-05,"completion":0.00017694639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.423659825329292e-06,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":46.38543525004487,"max_cost":46.38543525004487},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nex-agi/nex-n2-mini","alias_ids":null,"forwarded_model_id":null},{"id":"fugu-ultra","name":"Sakana: Fugu Ultra","created":1782276303,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":5.75,"max_completion_cost":4.4159999999999995,"max_cost":9.43},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.0,"max_prompt_cost":8847.319650658585,"max_completion_cost":6794.741491705791,"max_cost":14509.604227080077},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"sakana/fugu-ultra-20260615","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","created":1781754065,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":3.45e-06,"request":0.0,"image":0.0,"web_search":0.0161,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07536639999999999,"max_completion_cost":0.1130496,"max_cost":0.1695744},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.00530839179039515,"request":0.001,"image":0.0,"web_search":24.772495021844033,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":115.96358812511217,"max_completion_cost":173.9453821876683,"max_cost":260.9180732815024},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","created":1781754054,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.38e-05,"request":0.0,"image":2.2999999999999996e-06,"web_search":0.0161,"internal_reasoning":1.38e-05,"input_cache_read":2.2999999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.4521984,"max_cost":0.5275648},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0212335671615806,"request":0.001,"image":0.003538927860263433,"web_search":24.772495021844033,"internal_reasoning":0.0212335671615806,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0006635489737993938,"max_prompt_cost":231.92717625022433,"max_completion_cost":695.7815287506731,"max_cost":811.7451168757852},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3-pro-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3667e-07,"completion":7.438199999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.3952999999999997e-08,"input_cache_write":0.0,"max_prompt_cost":0.24235008,"max_completion_cost":0.09520895999999998,"max_cost":0.30726528},"sats_pricing":{"prompt":0.0003641556768211073,"completion":0.0011444892700091943,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.762891140963422e-05,"input_cache_write":0.0,"max_prompt_cost":372.8954130648139,"max_completion_cost":146.49462656117686,"max_cost":472.778112992889},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2:batch","name":"Z.ai: GLM 5.2 (batch)","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":512000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.049999999999999e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4949999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.41215999999999997,"max_completion_cost":1.29536,"max_cost":1.29536},"sats_pricing":{"prompt":0.0012386247510922017,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023003031091712315,"input_cache_write":0.0,"max_prompt_cost":634.1758725592073,"max_completion_cost":1993.124170900366,"max_cost":1993.124170900366},"per_request_limits":null,"top_provider":{"context_length":512000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"fusion","name":"OpenRouter: Fusion","created":1781371647,"description":"Fusion turns your prompt into a small multi-model deliberation. A panel of expert models (see below) analyzes your prompt in parallel with web search and web fetch enabled, then a...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":-1.15,"completion":-1.15,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-1150000.0,"max_completion_cost":-1150000.0,"max_cost":-1150000.0},"sats_pricing":{"prompt":-1769.4639301317166,"completion":-1769.4639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-1769463930.1317167,"max_completion_cost":-1769463930.1317167,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/fusion","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code","name":"MoonshotAI: Kimi K2.7 Code","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.049999999999999e-07,"completion":4.025e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":0.21102591999999998,"max_completion_cost":1.0551296,"max_cost":1.0551296},"sats_pricing":{"prompt":0.0012386247510922017,"completion":0.006193123755461008,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":324.6980467503141,"max_completion_cost":1623.4902337515705,"max_cost":1623.4902337515705},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code:batch","name":"MoonshotAI: Kimi K2.7 Code (batch)","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.4625e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0925e-07,"input_cache_write":0.0,"max_prompt_cost":0.14319616,"max_completion_cost":0.6029311999999999,"max_cost":0.6029311999999999},"sats_pricing":{"prompt":0.0008404953668125654,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001680990733625131,"input_cache_write":0.0,"max_prompt_cost":220.33081743771314,"max_completion_cost":927.7087050008973,"max_cost":927.7087050008973},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-latest","name":"Anthropic: Claude Fable Latest","created":1781029944,"description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":5.7499999999999995e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-06,"input_cache_write":1.4374999999999999e-05,"max_prompt_cost":11.5,"max_completion_cost":7.359999999999999,"max_cost":17.387999999999998},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.08847319650658583,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0017694639301317164,"input_cache_write":0.022118299126646458,"max_prompt_cost":17694.63930131717,"max_completion_cost":11324.569152842987,"max_cost":26754.294623591555},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~anthropic/claude-fable-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5","name":"Anthropic: Claude Fable 5","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":5.7499999999999995e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-06,"input_cache_write":1.4374999999999999e-05,"max_prompt_cost":11.5,"max_completion_cost":7.359999999999999,"max_cost":17.387999999999998},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.08847319650658583,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0017694639301317164,"input_cache_write":0.022118299126646458,"max_prompt_cost":17694.63930131717,"max_completion_cost":11324.569152842987,"max_cost":26754.294623591555},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5:batch","name":"Anthropic: Claude Fable 5 (batch)","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","created":1780937140,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.07536639999999999,"max_completion_cost":0.30146559999999994,"max_cost":0.30146559999999994},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":115.96358812511217,"max_completion_cost":463.85435250044867,"max_cost":463.85435250044867},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nex-agi/nex-n2-pro","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":4.139999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.35347871999999997,"max_completion_cost":2.1208723199999997,"max_cost":2.1208723199999997},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.006370070148474179,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":543.8850827035901,"max_completion_cost":3263.3104962215407,"max_cost":3263.3104962215407},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b:batch","name":"NVIDIA: Nemotron 3 Ultra (batch)","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.0699999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.17673935999999998,"max_completion_cost":1.0604361599999999,"max_cost":1.0604361599999999},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0031850350742370897,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":271.94254135179506,"max_completion_cost":1631.6552481107703,"max_cost":1631.6552481107703},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.6799999999999996e-07,"completion":1.4719999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.36e-08,"input_cache_write":4.5999999999999994e-07,"max_prompt_cost":0.36799999999999994,"max_completion_cost":0.19293798399999998,"max_cost":0.512703488},"sats_pricing":{"prompt":0.0005662284576421493,"completion":0.0022649138305685973,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011324569152842987,"input_cache_write":0.0007077855720526866,"max_prompt_cost":566.2284576421492,"max_completion_cost":296.8667856002872,"max_cost":788.8785468423647},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.7-plus-20260602","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.899999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":0.70656,"max_cost":0.71079936},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010616783580790298,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":1087.1586386729268,"max_cost":1093.6815905049643},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3:batch","name":"MiniMax: MiniMax M3 (batch)","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":524288,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.09043968,"max_completion_cost":0.36175872,"max_cost":0.36175872},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":139.15630575013463,"max_completion_cost":556.6252230005385,"max_cost":556.6252230005385},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.7-flash","name":"StepFun: Step 3.7 Flash","created":1779985069,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":1.3224999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5999999999999995e-08,"input_cache_write":0.0,"max_prompt_cost":0.058879999999999995,"max_completion_cost":0.33855999999999997,"max_cost":0.33855999999999997},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.002034883519651474,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.077855720526867e-05,"input_cache_write":0.0,"max_prompt_cost":90.59655322274389,"max_completion_cost":520.9301810307774,"max_cost":520.9301810307774},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":256000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"stepfun/step-3.7-flash-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","created":1779913703,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":5.7499999999999995e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-06,"input_cache_write":1.4374999999999999e-05,"max_prompt_cost":11.5,"max_completion_cost":7.359999999999999,"max_cost":17.387999999999998},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.08847319650658583,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0017694639301317164,"input_cache_write":0.022118299126646458,"max_prompt_cost":17694.63930131717,"max_completion_cost":11324.569152842987,"max_cost":26754.294623591555},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.8-opus-fast-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8","name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8:batch","name":"Anthropic: Claude Opus 4.8 (batch)","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":3.5937499999999997e-06,"max_prompt_cost":2.875,"max_completion_cost":1.8399999999999999,"max_cost":4.3469999999999995},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0055295747816616145,"max_prompt_cost":4423.659825329292,"max_completion_cost":2831.1422882107468,"max_cost":6688.573655897889},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-max","name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.6962499999999999e-06,"completion":5.088749999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3924999999999997e-07,"input_cache_write":2.1203125e-06,"max_prompt_cost":1.6962499999999998,"max_completion_cost":0.6669926399999999,"max_cost":2.14091176},"sats_pricing":{"prompt":0.002609959296944282,"completion":0.007829877890832846,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005219918593888564,"input_cache_write":0.003262449121180353,"max_prompt_cost":2609.959296944282,"max_completion_cost":1026.2777549072428,"max_cost":3294.144466882444},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.7-max-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"grok-build-0.1","name":"SpaceXAI: Grok Build 0.1","created":1779298123,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.29439999999999994,"max_completion_cost":0.5887999999999999,"max_cost":0.5887999999999999},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":452.9827661137194,"max_completion_cost":905.9655322274388,"max_cost":905.9655322274388},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"x-ai/grok-build-0.1-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash","name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":1.035e-05,"request":0.0,"image":1.725e-06,"web_search":0.0161,"internal_reasoning":1.035e-05,"input_cache_read":1.725e-07,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":1.8087936,"max_completion_cost":0.6782976,"max_cost":2.3740416},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.01592517537118545,"request":0.001,"image":0.002654195895197575,"web_search":24.772495021844033,"internal_reasoning":0.01592517537118545,"input_cache_read":0.00026541958951975753,"input_cache_write":0.00014745532751097632,"max_prompt_cost":2783.1261150026926,"max_completion_cost":1043.6722931260097,"max_cost":3652.853025941034},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash:batch","name":"Google: Gemini 3.5 Flash (batch)","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":8.625e-07,"completion":5.175e-06,"request":0.0,"image":8.625e-07,"web_search":0.0161,"internal_reasoning":5.175e-06,"input_cache_read":8.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.9043968,"max_completion_cost":0.3391488,"max_cost":1.1870208},"sats_pricing":{"prompt":0.0013270979475987876,"completion":0.007962587685592725,"request":0.001,"image":0.0013270979475987876,"web_search":24.772495021844033,"internal_reasoning":0.007962587685592725,"input_cache_read":0.00013270979475987876,"input_cache_write":0.0,"max_prompt_cost":1391.5630575013463,"max_completion_cost":521.8361465630048,"max_cost":1826.426512970517},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.45e-05,"completion":0.00017249999999999996,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.45e-06,"input_cache_write":4.312499999999999e-05,"max_prompt_cost":34.5,"max_completion_cost":22.079999999999995,"max_cost":52.163999999999994},"sats_pricing":{"prompt":0.0530839179039515,"completion":0.26541958951975747,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00530839179039515,"input_cache_write":0.06635489737993937,"max_prompt_cost":53083.9179039515,"max_completion_cost":33973.70745852895,"max_cost":80262.88387077466},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"perceptron-mk1","name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.725e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00565248,"max_completion_cost":0.0141312,"max_cost":0.01837056},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.002654195895197575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.697269109383415,"max_completion_cost":21.743172773458536,"max_cost":28.2661246054961},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perceptron/perceptron-mk1-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.625e-08,"completion":7.1875e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.02260992,"max_completion_cost":0.047104,"max_cost":0.06406144},"sats_pricing":{"prompt":0.00013270979475987876,"completion":0.001105914956332323,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":34.78907643753366,"max_completion_cost":72.47724257819512,"max_cost":98.56904990634536},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"inclusionai/ring-2.6-1t-20260508","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.725e-06,"request":0.0,"image":2.8749999999999995e-07,"web_search":0.0161,"internal_reasoning":1.725e-06,"input_cache_read":2.8749999999999996e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.30146559999999994,"max_completion_cost":0.1130496,"max_cost":0.39567359999999996},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.002654195895197575,"request":0.001,"image":0.0004423659825329291,"web_search":24.772495021844033,"internal_reasoning":0.002654195895197575,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":463.85435250044867,"max_completion_cost":173.9453821876683,"max_cost":608.8088376568389},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite:batch","name":"Google: Gemini 3.1 Flash Lite (batch)","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.4374999999999997e-07,"completion":8.625e-07,"request":0.0,"image":1.4374999999999997e-07,"web_search":0.0161,"internal_reasoning":8.625e-07,"input_cache_read":1.4374999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.0565248,"max_cost":0.19783679999999998},"sats_pricing":{"prompt":0.00022118299126646455,"completion":0.0013270979475987876,"request":0.001,"image":0.00022118299126646455,"web_search":24.772495021844033,"internal_reasoning":0.0013270979475987876,"input_cache_read":2.2118299126646457e-05,"input_cache_write":0.0,"max_prompt_cost":231.92717625022433,"max_completion_cost":86.97269109383414,"max_cost":304.40441882841947},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-chat-latest","name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":2.3,"max_completion_cost":4.4159999999999995,"max_cost":5.9799999999999995},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.0,"max_prompt_cost":3538.927860263433,"max_completion_cost":6794.741491705791,"max_cost":9201.212436684926},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-chat-latest-20260505","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.3","name":"SpaceXAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":1.4375,"max_completion_cost":2.875,"max_cost":2.875},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":2211.829912664646,"max_completion_cost":4423.659825329292,"max_cost":4423.659825329292},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"x-ai/grok-4.3-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":1.1499999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.015073279999999998,"max_cost":0.015073279999999998},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.00017694639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":23.192717625022436,"max_cost":23.192717625022436},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":8.625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.4521984,"max_completion_cost":2.260992,"max_cost":2.260992},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.013270979475987875,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":695.7815287506731,"max_completion_cost":3478.9076437533654,"max_cost":3478.9076437533654},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-medium-3.5-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-latest","name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":1.4375e-06,"max_prompt_cost":0.22999999999999995,"max_completion_cost":0.368,"max_cost":0.5244},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.002211829912664646,"max_prompt_cost":353.8927860263433,"max_completion_cost":566.2284576421494,"max_cost":806.8755521400628},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~anthropic/claude-haiku-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-mini-latest","name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":8.625e-07,"completion":5.175e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":8.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.345,"max_completion_cost":0.6624,"max_cost":0.897},"sats_pricing":{"prompt":0.0013270979475987876,"completion":0.007962587685592725,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00013270979475987876,"input_cache_write":0.0,"max_prompt_cost":530.839179039515,"max_completion_cost":1019.2112237558689,"max_cost":1380.1818655027391},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~openai/gpt-mini-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-pro-latest","name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.38e-05,"request":0.0,"image":2.2999999999999996e-06,"web_search":0.0161,"internal_reasoning":1.38e-05,"input_cache_read":2.2999999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":2.4117247999999996,"max_completion_cost":0.9043968,"max_cost":3.1653887999999997},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0212335671615806,"request":0.001,"image":0.003538927860263433,"web_search":24.772495021844033,"internal_reasoning":0.0212335671615806,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0006635489737993938,"max_prompt_cost":3710.8348200035894,"max_completion_cost":1391.5630575013463,"max_cost":4870.4707012547115},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~google/gemini-pro-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-latest","name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3349999999999996e-07,"input_cache_write":0.0,"max_prompt_cost":3.014656,"max_completion_cost":16.8820736,"max_cost":16.8820736},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005131445397381978,"input_cache_write":0.0,"max_prompt_cost":4638.5435250044875,"max_completion_cost":25975.843740025128,"max_cost":25975.843740025128},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":1048576,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~moonshotai/kimi-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-flash-latest","name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":8.625e-06,"request":0.0,"image":1.725e-06,"web_search":0.0161,"internal_reasoning":8.625e-06,"input_cache_read":1.725e-07,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":1.8087936,"max_completion_cost":0.565248,"max_cost":2.260992},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.013270979475987875,"request":0.001,"image":0.002654195895197575,"web_search":24.772495021844033,"internal_reasoning":0.013270979475987875,"input_cache_read":0.00026541958951975753,"input_cache_write":0.00014745532751097632,"max_prompt_cost":2783.1261150026926,"max_completion_cost":869.7269109383413,"max_cost":3478.9076437533654},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~google/gemini-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":2.875e-06,"max_prompt_cost":2.2999999999999994,"max_completion_cost":1.472,"max_cost":3.4776},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.004423659825329292,"max_prompt_cost":3538.9278602634326,"max_completion_cost":2264.9138305685974,"max_cost":5350.858924718311},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~anthropic/claude-sonnet-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-latest","name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":6.0375,"max_completion_cost":4.4159999999999995,"max_cost":9.7175},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":9289.685633191513,"max_completion_cost":6794.741491705791,"max_cost":14951.970209613006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~openai/gpt-latest","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.0699999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.3125e-07,"max_prompt_cost":0.345,"max_completion_cost":0.13565951999999998,"max_cost":0.45804959999999995},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0031850350742370897,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0006635489737993938,"max_prompt_cost":530.839179039515,"max_completion_cost":208.7344586252019,"max_cost":704.7845612271832},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-plus-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-flash","name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.15625e-07,"completion":1.29375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.6953125e-07,"max_prompt_cost":0.215625,"max_completion_cost":0.0847872,"max_cost":0.286281},"sats_pricing":{"prompt":0.0003317744868996969,"completion":0.0019906469213981813,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0004147181086246211,"max_prompt_cost":331.7744868996969,"max_completion_cost":130.4590366407512,"max_cost":440.4903507669896},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.6-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.30146559999999994,"max_cost":0.30146559999999994},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":463.85435250044867,"max_cost":463.85435250044867},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-max-preview","name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.18105e-06,"completion":7.086299999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":1.4763125e-06,"max_prompt_cost":0.3096051712,"max_completion_cost":0.46440775679999996,"max_cost":0.6966116351999999},"sats_pricing":{"prompt":0.001817239456245273,"completion":0.010903436737471638,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0022715493203065915,"max_prompt_cost":476.3784200179608,"max_completion_cost":714.5676300269413,"max_cost":1071.8514450404118},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.6-max-preview-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-27b","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":4.139999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3799999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":1.0852761599999998,"max_cost":1.0852761599999998},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.006370070148474179,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021233567161580596,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":1669.8756690016153,"max_cost":1669.8756690016153},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.6-27b-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.45e-05,"completion":0.000207,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":36.225,"max_completion_cost":26.496,"max_cost":58.30499999999999},"sats_pricing":{"prompt":0.0530839179039515,"completion":0.31850350742370903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55738.11379914908,"max_completion_cost":40768.44895023475,"max_cost":89711.82125767804},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro:batch","name":"OpenAI: GPT-5.5 Pro (batch)","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":0.0001035,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.1125,"max_completion_cost":13.248,"max_cost":29.152499999999996},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.15925175371185452,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27869.05689957454,"max_completion_cost":20384.224475117375,"max_cost":44855.91062883902},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":6.0375,"max_completion_cost":4.4159999999999995,"max_cost":9.7175},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.0,"max_prompt_cost":9289.685633191513,"max_completion_cost":6794.741491705791,"max_cost":14951.970209613006},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5:batch","name":"OpenAI: GPT-5.5 (batch)","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":3.01875,"max_completion_cost":2.2079999999999997,"max_cost":4.85875},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":4644.842816595757,"max_completion_cost":3397.3707458528957,"max_cost":7475.985104806503},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":5.0025e-07,"completion":1.0005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.16875e-09,"input_cache_write":0.0,"max_prompt_cost":0.524550144,"max_completion_cost":0.384192,"max_cost":0.716646144},"sats_pricing":{"prompt":0.0007697168096072968,"completion":0.0015394336192145937,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.414306746727473e-06,"input_cache_write":0.0,"max_prompt_cost":807.1065733507809,"max_completion_cost":591.1425097784039,"max_cost":1102.6778282399828},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v4-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.61e-07,"completion":3.22e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.22e-08,"input_cache_write":0.0,"max_prompt_cost":0.168820736,"max_completion_cost":0.126615552,"max_cost":0.232128512},"sats_pricing":{"prompt":0.00024772495021844034,"completion":0.0004954499004368807,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.954499004368806e-05,"input_cache_write":0.0,"max_prompt_cost":259.7584374002513,"max_completion_cost":194.81882805018847,"max_cost":357.1678514253456},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v4-flash-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.625e-08,"completion":7.1875e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.02260992,"max_completion_cost":0.023552,"max_cost":0.04333568},"sats_pricing":{"prompt":0.00013270979475987876,"completion":0.001105914956332323,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":34.78907643753366,"max_completion_cost":36.23862128909756,"max_cost":66.67906317193952},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"inclusionai/ling-2.6-1t-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.244999999999999e-08,"completion":2.415e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4149999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.018992332799999997,"max_completion_cost":0.063307776,"max_cost":0.063307776},"sats_pricing":{"prompt":0.00011147622759829815,"completion":0.0003715874253276605,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.715874253276605e-05,"input_cache_write":0.0,"max_prompt_cost":29.22282420752827,"max_completion_cost":97.40941402509424,"max_cost":97.40941402509424},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"tencent/hy3-preview-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.0025e-07,"completion":1.0005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.139999999999999e-09,"input_cache_write":0.0,"max_prompt_cost":0.524550144,"max_completion_cost":0.131137536,"max_cost":0.590118912},"sats_pricing":{"prompt":0.0007697168096072968,"completion":0.0015394336192145937,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.370070148474179e-06,"input_cache_write":0.0,"max_prompt_cost":807.1065733507809,"max_completion_cost":201.77664333769522,"max_cost":907.9948950196284},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.61e-07,"completion":3.22e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.2199999999999996e-09,"input_cache_write":0.0,"max_prompt_cost":0.168820736,"max_completion_cost":0.042205184,"max_cost":0.189923328},"sats_pricing":{"prompt":0.00024772495021844034,"completion":0.0004954499004368807,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.9544990043688066e-06,"input_cache_write":0.0,"max_prompt_cost":259.7584374002513,"max_completion_cost":64.93960935006282,"max_cost":292.2282420752827},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"xiaomi/mimo-v2.5-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.199999999999998e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.2999999999999996e-06,"input_cache_write":0.0,"max_prompt_cost":2.5023999999999997,"max_completion_cost":2.2079999999999997,"max_cost":3.5327999999999995},"sats_pricing":{"prompt":0.014155711441053731,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.003538927860263433,"input_cache_write":0.0,"max_prompt_cost":3850.3535119666153,"max_completion_cost":3397.3707458528957,"max_cost":5435.793193364633},"per_request_limits":null,"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-image-2-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999999e-08,"completion":3.449999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3e-09,"input_cache_write":0.0,"max_prompt_cost":0.0030146559999999997,"max_completion_cost":0.0011304959999999997,"max_cost":0.0037683199999999995},"sats_pricing":{"prompt":1.7694639301317167e-05,"completion":5.308391790395149e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.5389278602634336e-06,"input_cache_write":0.0,"max_prompt_cost":4.638543525004487,"max_completion_cost":1.7394538218766824,"max_cost":5.798179406255609},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"inclusionai/ling-2.6-flash-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-latest","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"~anthropic/claude-opus-latest","alias_ids":null,"forwarded_model_id":null},{"id":"pareto-code","name":"Pareto Code Router","created":1776747900,"description":"The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...","context_length":2000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":-1.15,"completion":-1.15,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-2300000.0,"max_completion_cost":-2300000.0,"max_cost":-2300000.0},"sats_pricing":{"prompt":-1769.4639301317166,"completion":-1769.4639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-3538927860.2634335,"max_completion_cost":-3538927860.2634335,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/pareto-code","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.664249999999999e-07,"completion":2.806e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1223999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":0.17469931519999998,"max_completion_cost":0.735576064,"max_cost":0.735576064},"sats_pricing":{"prompt":0.0010254043475113298,"completion":0.004317491989521389,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017269967958085555,"input_cache_write":0.0,"max_prompt_cost":268.80359727401003,"max_completion_cost":1131.804620101095,"max_cost":1131.804620101095},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2.6-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7","name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7:batch","name":"Anthropic: Claude Opus 4.7 (batch)","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":3.5937499999999997e-06,"max_prompt_cost":2.875,"max_completion_cost":1.8399999999999999,"max_cost":4.3469999999999995},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0055295747816616145,"max_prompt_cost":4423.659825329292,"max_completion_cost":2831.1422882107468,"max_cost":6688.573655897889},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0947999999999999e-06,"completion":3.4407999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0331999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.22197288959999997,"max_completion_cost":0.45099253759999997,"max_cost":0.5294678016},"sats_pricing":{"prompt":0.0016845296614853942,"completion":0.005294236078954097,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003128412228472875,"input_cache_write":0.0,"max_prompt_cost":341.54175792548665,"max_completion_cost":693.9261113406714,"max_cost":814.6731974759443},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5.1-20260406","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":8.05e-08,"completion":3.9099999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021102592,"max_completion_cost":0.006406143999999999,"max_cost":0.026189824},"sats_pricing":{"prompt":0.00012386247510922017,"completion":0.0006016177362447836,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.46980467503141,"max_completion_cost":9.856904990634535,"max_cost":40.297346873476485},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-31b-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":3.9099999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.10249830399999998,"max_cost":0.10249830399999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0006016177362447836,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":157.71047985015255,"max_cost":157.71047985015255},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-4-31b-it-20260402","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-plus","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.7374999999999997e-07,"completion":2.2424999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.671874999999999e-07,"max_prompt_cost":0.37374999999999997,"max_completion_cost":0.14696447999999998,"max_cost":0.4962204},"sats_pricing":{"prompt":0.0005750757772928079,"completion":0.003450454663756847,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0007188447216160098,"max_prompt_cost":575.0757772928079,"max_completion_cost":226.12899684396874,"max_cost":763.5166079961152},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.6-plus-04-02","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.38e-06,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.7599999999999993e-07,"input_cache_write":0.0,"max_prompt_cost":0.27979776,"max_completion_cost":0.6029311999999999,"max_cost":0.7018495999999999},"sats_pricing":{"prompt":0.0021233567161580602,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004246713432316119,"input_cache_write":0.0,"max_prompt_cost":430.514820914479,"max_completion_cost":927.7087050008973,"max_cost":1079.910914415107},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5v-turbo-20260401","alias_ids":null,"forwarded_model_id":null},{"id":"trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.53e-07,"completion":9.775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.899999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.066322432,"max_completion_cost":0.25624576,"max_cost":0.25624576},"sats_pricing":{"prompt":0.0003892820646289777,"completion":0.0015040443406119592,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010616783580790298,"input_cache_write":0.0,"max_prompt_cost":102.04795755009873,"max_completion_cost":394.2761996253814,"max_cost":394.2761996253814},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"arcee-ai/trinity-large-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":2.875,"max_completion_cost":5.75,"max_cost":5.75},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":4423.659825329292,"max_completion_cost":8847.319650658585,"max_cost":8847.319650658585},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20","name":"SpaceXAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":2.875,"max_completion_cost":5.75,"max_cost":5.75},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":4423.659825329292,"max_completion_cost":8847.319650658585,"max_cost":8847.319650658585},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"x-ai/grok-4.20-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-pro-preview","name":"Google: Lyria 3 Pro Preview","created":1774907286,"description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/lyria-3-pro-preview-20260330","alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-clip-preview","name":"Google: Lyria 3 Clip Preview","created":1774907255,"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/lyria-3-clip-preview-20260330","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.899999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.08832,"max_completion_cost":0.1104,"max_cost":0.17112},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010616783580790298,"input_cache_write":0.0,"max_prompt_cost":135.89482983411585,"max_completion_cost":169.86853729264482,"max_cost":263.2962328035994},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","alias_ids":null,"forwarded_model_id":null},{"id":"reka-edge","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":1.1499999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0018841599999999997,"max_completion_cost":0.0018841599999999997,"max_cost":0.0018841599999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.00017694639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8990897031278045,"max_completion_cost":2.8990897031278045,"max_cost":2.8990897031278045},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"rekaai/reka-edge-2603","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.899999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.070656,"max_completion_cost":0.18087936,"max_cost":0.20631551999999997},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010616783580790298,"input_cache_write":0.0,"max_prompt_cost":108.71586386729267,"max_completion_cost":278.31261150026927,"max_cost":317.4503224924946},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m2.7-20260318","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":1.4375e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.2999999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.09199999999999998,"max_completion_cost":0.184,"max_cost":0.24656},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.002211829912664646,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":3.538927860263433e-05,"input_cache_write":0.0,"max_prompt_cost":141.5571144105373,"max_completion_cost":283.1142288210747,"max_cost":379.3730666202401},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano:batch","name":"OpenAI: GPT-5.4 Nano (batch)","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":7.1875e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.04599999999999999,"max_completion_cost":0.092,"max_cost":0.12328},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.001105914956332323,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":70.77855720526865,"max_completion_cost":141.55711441053734,"max_cost":189.68653331012004},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.625e-07,"completion":5.175e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":8.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.345,"max_completion_cost":0.6624,"max_cost":0.897},"sats_pricing":{"prompt":0.0013270979475987876,"completion":0.007962587685592725,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00013270979475987876,"input_cache_write":0.0,"max_prompt_cost":530.839179039515,"max_completion_cost":1019.2112237558689,"max_cost":1380.1818655027391},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini:batch","name":"OpenAI: GPT-5.4 Mini (batch)","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.3125e-07,"completion":2.5875e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":4.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1725,"max_completion_cost":0.3312,"max_cost":0.4485},"sats_pricing":{"prompt":0.0006635489737993938,"completion":0.0039812938427963625,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":6.635489737993938e-05,"input_cache_write":0.0,"max_prompt_cost":265.4195895197575,"max_completion_cost":509.60561187793445,"max_cost":690.0909327513696},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.18087936,"max_cost":0.18087936},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":278.31261150026927,"max_cost":278.31261150026927},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-small-2603","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-turbo","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.38e-06,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.7599999999999993e-07,"input_cache_write":0.0,"max_prompt_cost":0.27979776,"max_completion_cost":0.6029311999999999,"max_cost":0.7018495999999999},"sats_pricing":{"prompt":0.0021233567161580602,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004246713432316119,"input_cache_write":0.0,"max_prompt_cost":430.514820914479,"max_completion_cost":927.7087050008973,"max_cost":1079.910914415107},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5-turbo-20260315","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.774999999999999e-08,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.025624575999999996,"max_completion_cost":0.007536639999999999,"max_cost":0.03155967999999999},"sats_pricing":{"prompt":0.0001504044340611959,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":39.42761996253814,"max_completion_cost":11.596358812511218,"max_cost":48.55975252739072},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07536639999999999,"max_completion_cost":0.30146559999999994,"max_cost":0.3391487999999999},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":115.96358812511217,"max_completion_cost":463.85435250044867,"max_cost":521.8361465630047},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-9b","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":1.725e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.04521984,"max_cost":0.04521984},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.00026541958951975753,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":69.57815287506732,"max_cost":69.57815287506732},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-9b-20260310","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.45e-05,"completion":0.000207,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":36.225,"max_completion_cost":26.496,"max_cost":58.30499999999999},"sats_pricing":{"prompt":0.0530839179039515,"completion":0.31850350742370903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":55738.11379914908,"max_completion_cost":40768.44895023475,"max_cost":89711.82125767804},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro:batch","name":"OpenAI: GPT-5.4 Pro (batch)","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":0.0001035,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.1125,"max_completion_cost":13.248,"max_cost":29.152499999999996},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.15925175371185452,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27869.05689957454,"max_completion_cost":20384.224475117375,"max_cost":44855.91062883902},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":3.01875,"max_completion_cost":2.2079999999999997,"max_cost":4.85875},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":4644.842816595757,"max_completion_cost":3397.3707458528957,"max_cost":7475.985104806503},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4:batch","name":"OpenAI: GPT-5.4 (batch)","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":8.625e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":1.509375,"max_completion_cost":1.1039999999999999,"max_cost":2.429375},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.013270979475987875,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0,"max_prompt_cost":2322.4214082978783,"max_completion_cost":1698.6853729264478,"max_cost":3737.9925524032515},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"mercury-2","name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":8.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.03679999999999999,"max_completion_cost":0.043125,"max_cost":0.06555},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0013270979475987876,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":56.62284576421492,"max_completion_cost":66.35489737993937,"max_cost":100.85944401750785},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"inception/mercury-2-20260304","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-chat","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.0125e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.0124999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.2576,"max_completion_cost":0.2637824,"max_cost":0.48840959999999994},"sats_pricing":{"prompt":0.003096561877730504,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003096561877730504,"input_cache_write":0.0,"max_prompt_cost":396.35992034950453,"max_completion_cost":405.8725584378926,"max_cost":751.4984089826605},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.3-chat-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.725e-06,"request":0.0,"image":2.8749999999999995e-07,"web_search":0.0161,"internal_reasoning":1.725e-06,"input_cache_read":2.8749999999999996e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.30146559999999994,"max_completion_cost":0.1130496,"max_cost":0.39567359999999996},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.002654195895197575,"request":0.001,"image":0.0004423659825329291,"web_search":24.772495021844033,"internal_reasoning":0.002654195895197575,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":463.85435250044867,"max_completion_cost":173.9453821876683,"max_cost":608.8088376568389},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.06029311999999999,"max_cost":0.07536639999999999},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":92.77087050008974,"max_cost":115.96358812511217},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":3.45e-06,"request":0.0,"image":0.0,"web_search":0.0161,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.03768319999999999,"max_completion_cost":0.2260992,"max_cost":0.2260992},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.00530839179039515,"request":0.001,"image":0.0,"web_search":24.772495021844033,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":57.981794062556084,"max_completion_cost":347.8907643753366,"max_cost":347.8907643753366},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.61e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.042205184,"max_completion_cost":0.30146559999999994,"max_cost":0.30146559999999994},"sats_pricing":{"prompt":0.00024772495021844034,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":64.93960935006282,"max_completion_cost":463.85435250044867,"max_cost":463.85435250044867},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-27b","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2424999999999999e-07,"completion":1.7939999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.058785791999999996,"max_completion_cost":0.11757158399999999,"max_cost":0.16166092799999998},"sats_pricing":{"prompt":0.00034504546637568475,"completion":0.002760363731005478,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":90.4515987375875,"max_completion_cost":180.903197475175,"max_cost":248.74189652836563},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-27b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.3349999999999996e-07,"completion":2.76e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08742502399999999,"max_completion_cost":0.2260992,"max_cost":0.286203904},"sats_pricing":{"prompt":0.0005131445397381978,"completion":0.0042467134323161205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":134.51776222513013,"max_completion_cost":347.8907643753366,"max_cost":440.3717259051136},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.474999999999999e-08,"completion":2.9899999999999996e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07475,"max_completion_cost":0.019595263999999998,"max_cost":0.089446448},"sats_pricing":{"prompt":0.00011501515545856157,"completion":0.0004600606218342463,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":115.01515545856158,"max_completion_cost":30.150532912529165,"max_cost":137.62805514295846},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-flash-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.38e-05,"request":0.0,"image":2.2999999999999996e-06,"web_search":0.0161,"internal_reasoning":1.38e-05,"input_cache_read":2.2999999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":2.4117247999999996,"max_completion_cost":0.9043968,"max_cost":3.1653887999999997},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0212335671615806,"request":0.001,"image":0.003538927860263433,"web_search":24.772495021844033,"internal_reasoning":0.0212335671615806,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0006635489737993938,"max_prompt_cost":3710.8348200035894,"max_completion_cost":1391.5630575013463,"max_cost":4870.4707012547115},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-codex","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.0125e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.0124999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.8049999999999999,"max_completion_cost":2.0608,"max_cost":2.6082},"sats_pricing":{"prompt":0.003096561877730504,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003096561877730504,"input_cache_write":0.0,"max_prompt_cost":1238.6247510922017,"max_completion_cost":3170.8793627960363,"max_cost":4013.144193538734},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.3-codex-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"aion-2.0","name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.199999999999999e-07,"completion":1.8399999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.12058623999999998,"max_completion_cost":0.06029311999999999,"max_cost":0.1507328},"sats_pricing":{"prompt":0.0014155711441053733,"completion":0.0028311422882107465,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":185.54174100017948,"max_completion_cost":92.77087050008974,"max_cost":231.9271762502244},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"aion-labs/aion-2.0-20260223","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.38e-05,"request":0.0,"image":2.2999999999999996e-06,"web_search":0.0161,"internal_reasoning":1.38e-05,"input_cache_read":2.2999999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":2.4117247999999996,"max_completion_cost":0.9043968,"max_cost":3.1653887999999997},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0212335671615806,"request":0.001,"image":0.003538927860263433,"web_search":24.772495021844033,"internal_reasoning":0.0212335671615806,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0006635489737993938,"max_prompt_cost":3710.8348200035894,"max_completion_cost":1391.5630575013463,"max_cost":4870.4707012547115},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview:batch","name":"Google: Gemini 3.1 Pro Preview (batch)","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":6.9e-06,"request":0.0,"image":1.1499999999999998e-06,"web_search":0.0161,"internal_reasoning":6.9e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.2058623999999998,"max_completion_cost":0.4521984,"max_cost":1.5826943999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0106167835807903,"request":0.001,"image":0.0017694639301317164,"web_search":24.772495021844033,"internal_reasoning":0.0106167835807903,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1855.4174100017947,"max_completion_cost":695.7815287506731,"max_cost":2435.2353506273557},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":4.3125e-06,"max_prompt_cost":3.45,"max_completion_cost":2.2079999999999997,"max_cost":5.2164},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.006635489737993937,"max_prompt_cost":5308.391790395151,"max_completion_cost":3397.3707458528957,"max_cost":8026.288387077468},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6:batch","name":"Anthropic: Claude Sonnet 4.6 (batch)","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":8.625e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":2.15625e-06,"max_prompt_cost":1.725,"max_completion_cost":1.1039999999999999,"max_cost":2.6082},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.013270979475987875,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0033177448689969686,"max_prompt_cost":2654.1958951975753,"max_completion_cost":1698.6853729264478,"max_cost":4013.144193538734},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.9899999999999996e-07,"completion":1.7939999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.299,"max_completion_cost":0.11757158399999999,"max_cost":0.39697632},"sats_pricing":{"prompt":0.0004600606218342463,"completion":0.002760363731005478,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":460.0606218342463,"max_completion_cost":180.903197475175,"max_cost":610.8132863968922},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-plus-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.4849999999999997e-07,"completion":2.6909999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11757158399999999,"max_completion_cost":0.17635737599999998,"max_cost":0.26453606399999996},"sats_pricing":{"prompt":0.0006900909327513695,"completion":0.004140545596508217,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":180.903197475175,"max_completion_cost":271.3547962127625,"max_cost":407.03219431914374},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.53e-07,"completion":1.0349999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.049741824000000004,"max_completion_cost":0.20348927999999997,"max_cost":0.20348927999999997},"sats_pricing":{"prompt":0.0003892820646289777,"completion":0.0015925175371185448,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.0,"max_prompt_cost":76.53596816257405,"max_completion_cost":313.10168793780286,"max_cost":313.10168793780286},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m2.5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0925e-06,"completion":2.9325e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.22374399999999997,"max_completion_cost":0.38436864,"max_cost":0.46491647999999997},"sats_pricing":{"prompt":0.0016809907336251307,"completion":0.0045121330218358775,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":344.26690224642675,"max_completion_cost":591.4142994380721,"max_cost":715.3503842467858},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":8.969999999999999e-07,"completion":4.484999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.23514316799999999,"max_completion_cost":0.29392895999999996,"max_cost":0.4702863359999999},"sats_pricing":{"prompt":0.001380181865502739,"completion":0.006900909327513694,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":361.80639495035,"max_completion_cost":452.25799368793747,"max_cost":723.6127899006999},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-max-thinking-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":5.75,"max_completion_cost":3.6799999999999997,"max_cost":8.693999999999999},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":8847.319650658585,"max_completion_cost":5662.2845764214935,"max_cost":13377.147311795778},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6:batch","name":"Anthropic: Claude Opus 4.6 (batch)","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":3.5937499999999997e-06,"max_prompt_cost":2.875,"max_completion_cost":1.8399999999999999,"max_cost":4.3469999999999995},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0055295747816616145,"max_prompt_cost":4423.659825329292,"max_completion_cost":2831.1422882107468,"max_cost":6688.573655897889},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.3799999999999997e-07,"completion":9.199999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.05e-08,"input_cache_write":0.0,"max_prompt_cost":0.03617587199999999,"max_completion_cost":0.24117247999999997,"max_cost":0.24117247999999997},"sats_pricing":{"prompt":0.00021233567161580596,"completion":0.0014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012386247510922017,"input_cache_write":0.0,"max_prompt_cost":55.66252230005384,"max_completion_cost":371.08348200035897,"max_cost":371.08348200035897},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","alias_ids":null,"forwarded_model_id":null},{"id":"free","name":"Free Models Router","created":1769917427,"description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/free","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.5-flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.02260992,"max_cost":0.04521984},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":34.78907643753366,"max_cost":69.57815287506732},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"stepfun/step-3.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.555e-07,"completion":3.2774999999999995e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0925e-07,"input_cache_write":0.0,"max_prompt_cost":0.171835392,"max_completion_cost":0.8591769599999999,"max_cost":0.8591769599999999},"sats_pricing":{"prompt":0.0010085944401750785,"completion":0.0050429722008753924,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001680990733625131,"input_cache_write":0.0,"max_prompt_cost":264.3969809252558,"max_completion_cost":1321.9849046262789,"max_cost":1321.9849046262789},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2.5-0127","alias_ids":null,"forwarded_model_id":null},{"id":"solar-pro-3","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.02260992,"max_completion_cost":0.09043968,"max_cost":0.09043968},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":34.78907643753366,"max_completion_cost":139.15630575013463,"max_cost":139.15630575013463},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"upstage/solar-pro-3","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2-her","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.02260992,"max_completion_cost":0.00282624,"max_cost":0.0247296},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":34.78907643753366,"max_completion_cost":4.348634554691707,"max_cost":38.05055235355244},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m2-her-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"palmyra-x5","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.7175999999999999,"max_completion_cost":0.0565248,"max_cost":0.76847232},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1104.1454924021912,"max_completion_cost":86.97269109383414,"max_cost":1182.420914386642},"per_request_limits":null,"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"writer/palmyra-x5-20250428","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.368,"max_completion_cost":0.188416,"max_cost":0.509312},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":566.2284576421494,"max_completion_cost":289.90897031278047,"max_cost":783.6601853767347},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-audio","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio-mini","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":2.76e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08832,"max_completion_cost":0.04521984,"max_cost":0.12223487999999999},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.0042467134323161205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":135.89482983411585,"max_completion_cost":69.57815287506732,"max_cost":188.07844449041633},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-audio-mini","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7-flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.899999999999998e-08,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.013989887999999997,"max_completion_cost":0.007536639999999999,"max_cost":0.020396031999999994},"sats_pricing":{"prompt":0.00010616783580790298,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":21.525741045723947,"max_completion_cost":11.596358812511218,"max_cost":31.382646036358476},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.7-flash-20260119","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-codex","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.0125e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.0124999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.8049999999999999,"max_completion_cost":2.0608,"max_cost":2.6082},"sats_pricing":{"prompt":0.003096561877730504,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003096561877730504,"input_cache_write":0.0,"max_prompt_cost":1238.6247510922017,"max_completion_cost":3170.8793627960363,"max_cost":4013.144193538734},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-codex-20260114","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.625e-08,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02260992,"max_completion_cost":0.01130496,"max_cost":0.031088639999999997},"sats_pricing":{"prompt":0.00013270979475987876,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.78907643753366,"max_completion_cost":17.39453821876683,"max_cost":47.83498010160878},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07536639999999999,"max_completion_cost":0.07536639999999999,"max_cost":0.141312},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":115.96358812511217,"max_completion_cost":115.96358812511217,"max_cost":217.43172773458534},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"bytedance-seed/seed-1.6-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.070656,"max_completion_cost":0.18087936,"max_cost":0.20631551999999997},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":108.71586386729267,"max_completion_cost":278.31261150026927,"max_cost":317.4503224924946},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m2.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":2.0125e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.199999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.09326591999999999,"max_completion_cost":0.2637824,"max_cost":0.2967552},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.003096561877730504,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00014155711441053733,"input_cache_write":0.0,"max_prompt_cost":143.5049403048263,"max_completion_cost":405.8725584378926,"max_cost":456.60662824262926},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.7-20251222","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":3.45e-06,"request":0.0,"image":5.749999999999999e-07,"web_search":0.0161,"internal_reasoning":3.45e-06,"input_cache_read":5.749999999999999e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.6029311999999999,"max_completion_cost":0.2260992,"max_cost":0.7913471999999999},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.00530839179039515,"request":0.001,"image":0.0008847319650658582,"web_search":24.772495021844033,"internal_reasoning":0.00530839179039515,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":927.7087050008973,"max_completion_cost":347.8907643753366,"max_cost":1217.6176753136779},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview:batch","name":"Google: Gemini 3 Flash Preview (batch)","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.725e-06,"request":0.0,"image":2.8749999999999995e-07,"web_search":0.0161,"internal_reasoning":1.725e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.30146559999999994,"max_completion_cost":0.1130496,"max_cost":0.39567359999999996},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.002654195895197575,"request":0.001,"image":0.0004423659825329291,"web_search":24.772495021844033,"internal_reasoning":0.002654195895197575,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":463.85435250044867,"max_completion_cost":173.9453821876683,"max_cost":608.8088376568389},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.015073279999999998,"max_completion_cost":0.06029311999999999,"max_cost":0.06029311999999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":23.192717625022436,"max_completion_cost":92.77087050008974,"max_cost":92.77087050008974},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.0125e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.0124999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.2576,"max_completion_cost":0.2637824,"max_cost":0.48840959999999994},"sats_pricing":{"prompt":0.003096561877730504,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003096561877730504,"input_cache_write":0.0,"max_prompt_cost":396.35992034950453,"max_completion_cost":405.8725584378926,"max_cost":751.4984089826605},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-chat-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.4149999999999997e-05,"completion":0.00019319999999999998,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.659999999999998,"max_completion_cost":24.729599999999998,"max_cost":31.298399999999997},"sats_pricing":{"prompt":0.03715874253276605,"completion":0.2972699402621284,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14863.497013106418,"max_completion_cost":38050.55235355243,"max_cost":48157.7303224648},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro:batch","name":"OpenAI: GPT-5.2 Pro (batch)","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.2074999999999999e-05,"completion":9.659999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.829999999999999,"max_completion_cost":12.364799999999999,"max_cost":15.649199999999999},"sats_pricing":{"prompt":0.018579371266383024,"completion":0.1486349701310642,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7431.748506553209,"max_completion_cost":19025.276176776217,"max_cost":24078.8651612324},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.0125e-06,"completion":1.61e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.0124999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.8049999999999999,"max_completion_cost":2.0608,"max_cost":2.6082},"sats_pricing":{"prompt":0.003096561877730504,"completion":0.024772495021844032,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0003096561877730504,"input_cache_write":0.0,"max_prompt_cost":1238.6247510922017,"max_completion_cost":3170.8793627960363,"max_cost":4013.144193538734},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2:batch","name":"OpenAI: GPT-5.2 (batch)","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.00625e-06,"completion":8.05e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.0062499999999999e-07,"input_cache_write":0.0,"max_prompt_cost":0.40249999999999997,"max_completion_cost":1.0304,"max_cost":1.3041},"sats_pricing":{"prompt":0.001548280938865252,"completion":0.012386247510922016,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0001548280938865252,"input_cache_write":0.0,"max_prompt_cost":619.3123755461008,"max_completion_cost":1585.4396813980181,"max_cost":2006.572096769367},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"relace-search","name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":3.45e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29439999999999994,"max_completion_cost":0.4416,"max_cost":0.5888},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.00530839179039515,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":452.9827661137194,"max_completion_cost":679.4741491705793,"max_cost":905.965532227439},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"relace/relace-search-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6v","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.0349999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.325e-08,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.033914879999999994,"max_cost":0.06782975999999999},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0015925175371185448,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.732051615724442e-05,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":52.18361465630048,"max_cost":104.36722931260095},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.6-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"bodybuilder","name":"Body Builder (beta)","created":1764903653,"description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":-1.15,"completion":-1.15,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-147200.0,"max_completion_cost":-147200.0,"max_cost":-147200.0},"sats_pricing":{"prompt":-1769.4639301317166,"completion":-1769.4639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-226491383.05685976,"max_completion_cost":-226491383.05685976,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/bodybuilder","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.575,"max_completion_cost":1.472,"max_cost":1.863},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0,"max_prompt_cost":884.7319650658583,"max_completion_cost":2264.9138305685974,"max_cost":2866.5315668133812},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.1-codex-max-20251204","alias_ids":null,"forwarded_model_id":null},{"id":"nova-2-lite-v1","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.345,"max_completion_cost":0.188413125,"max_cost":0.51080355},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":530.839179039515,"max_completion_cost":289.9045466529551,"max_cost":785.9551800941156},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"amazon/nova-2-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.06029311999999999,"max_cost":0.06029311999999999},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.538927860263433e-05,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":92.77087050008974,"max_cost":92.77087050008974},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/ministral-14b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.725e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7249999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.04521984,"max_cost":0.04521984},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.00026541958951975753,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6541958951975745e-05,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":69.57815287506732,"max_cost":69.57815287506732},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/ministral-8b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":1.1499999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.015073279999999998,"max_completion_cost":0.015073279999999998,"max_cost":0.015073279999999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.00017694639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":23.192717625022436,"max_completion_cost":23.192717625022436,"max_cost":23.192717625022436},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/ministral-3b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2512","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":1.725e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.4521984,"max_cost":0.4521984},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.002654195895197575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.0,"max_prompt_cost":231.92717625022433,"max_completion_cost":695.7815287506731,"max_cost":695.7815287506731},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-large-2512","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":3.0934999999999995e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5467499999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.05068390399999999,"max_completion_cost":0.030146559999999996,"max_cost":0.06055690239999999},"sats_pricing":{"prompt":0.0004759857972054317,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023799289860271586,"input_cache_write":0.0,"max_prompt_cost":77.98551301413794,"max_completion_cost":46.38543525004487,"max_cost":93.17674305852762},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v3.2-20251201","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":2.8749999999999997e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":7.187499999999999e-06,"max_prompt_cost":1.15,"max_completion_cost":1.8399999999999999,"max_cost":2.622},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.044236598253292916,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.011059149563323229,"max_prompt_cost":1769.4639301317166,"max_completion_cost":2831.1422882107468,"max_cost":4034.377760700314},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5:batch","name":"Anthropic: Claude Opus 4.5 (batch)","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":3.5937499999999997e-06,"max_prompt_cost":0.575,"max_completion_cost":0.9199999999999999,"max_cost":1.311},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0055295747816616145,"max_prompt_cost":884.7319650658583,"max_completion_cost":1415.5711441053734,"max_cost":2017.188880350157},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":5.749999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01130496,"max_completion_cost":0.03768319999999999,"max_cost":0.03768319999999999},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0008847319650658582,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17.39453821876683,"max_completion_cost":57.981794062556084,"max_cost":57.981794062556084},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"allenai/olmo-3-32b-think-20251121","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":1.38e-05,"request":0.0,"image":2.2999999999999996e-06,"web_search":0.0161,"internal_reasoning":1.38e-05,"input_cache_read":2.2999999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.4521984,"max_cost":0.5275648},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0212335671615806,"request":0.001,"image":0.003538927860263433,"web_search":24.772495021844033,"internal_reasoning":0.0212335671615806,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0006635489737993938,"max_prompt_cost":231.92717625022433,"max_completion_cost":695.7815287506731,"max_cost":811.7451168757852},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-3-pro-image-preview-20251120","alias_ids":null,"forwarded_model_id":null},{"id":"cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.4375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.184,"max_completion_cost":0.184,"max_cost":0.184},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.002211829912664646,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":283.1142288210747,"max_completion_cost":283.1142288210747,"max_cost":283.1142288210747},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.575,"max_completion_cost":1.472,"max_cost":1.863},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0,"max_prompt_cost":884.7319650658583,"max_completion_cost":2264.9138305685974,"max_cost":2866.5315668133812},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1:batch","name":"OpenAI: GPT-5.1 (batch)","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.1875e-07,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":7.187499999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.2875,"max_completion_cost":0.736,"max_cost":0.9315},"sats_pricing":{"prompt":0.001105914956332323,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00011059149563323228,"input_cache_write":0.0,"max_prompt_cost":442.36598253292914,"max_completion_cost":1132.4569152842987,"max_cost":1433.2657834066906},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4949999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.575,"max_completion_cost":1.472,"max_cost":1.863},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00023003031091712315,"input_cache_write":0.0,"max_prompt_cost":884.7319650658583,"max_completion_cost":2264.9138305685974,"max_cost":2866.5315668133812},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.1-codex-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.11499999999999998,"max_completion_cost":0.29439999999999994,"max_cost":0.37259999999999993},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":176.94639301317164,"max_completion_cost":452.9827661137194,"max_cost":573.3063133626762},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":0.288512,"max_cost":0.40014848},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":443.9231107914451,"max_cost":615.6941757017674},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","alias_ids":null,"forwarded_model_id":null},{"id":"nova-premier-v1","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.4374999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.1875e-07,"input_cache_write":0.0,"max_prompt_cost":2.875,"max_completion_cost":0.45999999999999996,"max_cost":3.243},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.022118299126646458,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.001105914956332323,"input_cache_write":0.0,"max_prompt_cost":4423.659825329292,"max_completion_cost":707.7855720526867,"max_cost":4989.888282971441},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"amazon/nova-premier-v1","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro-search","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.020699999999999996,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.69,"max_completion_cost":0.13799999999999998,"max_cost":0.8004},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":31.850350742370896,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1061.67835807903,"max_completion_cost":212.33567161580598,"max_cost":1231.546895371675},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perplexity/sonar-pro-search","alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.0036799999999999997,"max_completion_cost":0.01104,"max_cost":0.01104},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":5.662284576421493,"max_completion_cost":16.98685372926448,"max_cost":16.98685372926448},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/voxtral-small-24b-2507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.625e-08,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.01130496,"max_completion_cost":0.02260992,"max_cost":0.0282624},"sats_pricing":{"prompt":0.00013270979475987876,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.635489737993938e-05,"input_cache_write":0.0,"max_prompt_cost":17.39453821876683,"max_completion_cost":34.78907643753366,"max_cost":43.48634554691707},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-oss-safeguard-20b","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.9324999999999996e-07,"completion":1.1729999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06005759999999999,"max_completion_cost":0.15374745599999998,"max_cost":0.17536819199999998},"sats_pricing":{"prompt":0.00045121330218358773,"completion":0.001804853208734351,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":92.40848428719876,"max_completion_cost":236.56571977522884,"max_cost":269.8327741186204},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m2","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.1959999999999999e-07,"completion":4.783999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.015676211199999998,"max_completion_cost":0.015676211199999998,"max_cost":0.027433369599999997},"sats_pricing":{"prompt":0.0001840242487336985,"completion":0.000736096994934794,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24.12042633002333,"max_completion_cost":24.12042633002333,"max_cost":42.21074607754083},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.955e-08,"completion":1.288e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00256105,"max_completion_cost":0.016872799999999997,"max_cost":0.016872799999999997},"sats_pricing":{"prompt":3.0080886812239185e-05,"completion":0.00019817996017475225,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.940596172403333,"max_completion_cost":25.961574782892544,"max_cost":25.961574782892544},"per_request_limits":null,"top_provider":{"context_length":131000,"max_completion_tokens":131000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"ibm-granite/granite-4.0-h-micro","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":1.15,"max_completion_cost":0.29439999999999994,"max_cost":1.0764},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":1769.4639301317166,"max_completion_cost":452.9827661137194,"max_cost":1656.2182386032869},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-image-mini","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":1.4375e-06,"max_prompt_cost":0.22999999999999995,"max_completion_cost":0.368,"max_cost":0.5244},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.002211829912664646,"max_prompt_cost":353.8927860263433,"max_completion_cost":566.2284576421494,"max_cost":806.8755521400628},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5:batch","name":"Anthropic: Claude Haiku 4.5 (batch)","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":7.1875e-07,"max_prompt_cost":0.11499999999999998,"max_completion_cost":0.184,"max_cost":0.2622},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.001105914956332323,"max_prompt_cost":176.94639301317164,"max_completion_cost":283.1142288210747,"max_cost":403.4377760700314},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.07e-07,"completion":2.4149999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027131904,"max_completion_cost":0.07913471999999999,"max_cost":0.099483648},"sats_pricing":{"prompt":0.000318503507423709,"completion":0.003715874253276605,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":41.746891725040385,"max_completion_cost":121.7617675313678,"max_cost":153.07193632514807},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-8b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.3455e-07,"completion":5.232499999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0176357376,"max_completion_cost":0.017145855999999998,"max_cost":0.0303726592},"sats_pricing":{"prompt":0.00020702727982541087,"completion":0.000805106088209931,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.135479621276254,"max_completion_cost":26.38171629846302,"max_cost":46.733326014420214},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4375e-06,"input_cache_write":0.0,"max_prompt_cost":4.6,"max_completion_cost":1.472,"max_cost":4.6},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.002211829912664646,"input_cache_write":0.0,"max_prompt_cost":7077.855720526866,"max_completion_cost":2264.9138305685974,"max_cost":7077.855720526866},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-image","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.875e-06,"request":0.0,"image":3.45e-07,"web_search":0.0161,"internal_reasoning":2.875e-06,"input_cache_read":3.449999999999999e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.01130496,"max_completion_cost":0.023552,"max_cost":0.03203072},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.004423659825329292,"request":0.001,"image":0.0005308391790395151,"web_search":24.772495021844033,"internal_reasoning":0.004423659825329292,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":17.39453821876683,"max_completion_cost":36.23862128909756,"max_cost":49.28452495317268},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-flash-image","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":2.76e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.09043968,"max_cost":0.1130496},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0042467134323161205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":139.15630575013463,"max_cost":173.9453821876683},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.01130496,"max_cost":0.05369855999999999},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":17.39453821876683,"max_cost":82.62405653914243},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":0.000138,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.8999999999999995,"max_completion_cost":17.663999999999998,"max_cost":22.355999999999998},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.212335671615806,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10616.7835807903,"max_completion_cost":27178.965966823165,"max_cost":34398.37880176057},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro:batch","name":"OpenAI: GPT-5 Pro (batch)","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.625e-06,"completion":6.9e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.4499999999999997,"max_completion_cost":8.831999999999999,"max_cost":11.177999999999999},"sats_pricing":{"prompt":0.013270979475987875,"completion":0.106167835807903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5308.39179039515,"max_completion_cost":13589.482983411583,"max_cost":17199.189400880285},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.11658239999999997,"max_completion_cost":0.30146559999999994,"max_cost":0.3426815999999999},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":179.38117538103288,"max_completion_cost":463.85435250044867,"max_cost":527.2719397563694},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.6","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":4.3125e-06,"max_prompt_cost":3.45,"max_completion_cost":1.1039999999999999,"max_cost":4.3332},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.006635489737993937,"max_prompt_cost":5308.391790395151,"max_completion_cost":1698.6853729264478,"max_cost":6667.340088736309},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5:batch","name":"Anthropic: Claude Sonnet 4.5 (batch)","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.725e-06,"completion":8.625e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":2.15625e-06,"max_prompt_cost":1.725,"max_completion_cost":0.5519999999999999,"max_cost":2.1666},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.013270979475987875,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0033177448689969686,"max_prompt_cost":2654.1958951975753,"max_completion_cost":849.3426864632239,"max_cost":3333.6700443681543},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":3.105e-07,"completion":4.7149999999999995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05087232,"max_completion_cost":0.030900223999999997,"max_cost":0.06142361599999999},"sats_pricing":{"prompt":0.0004777552611355635,"completion":0.0007254802113540038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":78.27542198445073,"max_completion_cost":47.54507113129599,"max_cost":94.51032432196642},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v3.2-exp","alias_ids":null,"forwarded_model_id":null},{"id":"cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":5.749999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.725e-07,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.07536639999999999,"max_cost":0.07536639999999999},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0008847319650658582,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00026541958951975753,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":115.96358812511217,"max_cost":115.96358812511217},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thedrummer/cydonia-24b-v4.1","alias_ids":null,"forwarded_model_id":null},{"id":"relace-apply-3","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.775e-07,"completion":1.4375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.25023999999999996,"max_completion_cost":0.184,"max_cost":0.30911999999999995},"sats_pricing":{"prompt":0.0015040443406119592,"completion":0.002211829912664646,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":385.0353511966615,"max_completion_cost":283.1142288210747,"max_cost":475.6319044194054},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"relace/relace-apply-3","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.15073279999999997,"max_cost":0.19595263999999996},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":231.92717625022433,"max_cost":301.5053291252916},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.415e-07,"completion":2.185e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.031653888,"max_completion_cost":0.07159808,"max_cost":0.095338496},"sats_pricing":{"prompt":0.0003715874253276605,"completion":0.0033619814672502615,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":48.70470701254712,"max_completion_cost":110.16540871885657,"max_cost":146.6939389782669},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.969999999999999e-07,"completion":4.484999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7939999999999998e-07,"input_cache_write":1.1212499999999999e-06,"max_prompt_cost":0.23514316799999999,"max_completion_cost":0.29392895999999996,"max_cost":0.4702863359999999},"sats_pricing":{"prompt":0.001380181865502739,"completion":0.006900909327513694,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002760363731005478,"input_cache_write":0.0017252273318784236,"max_prompt_cost":361.80639495035,"max_completion_cost":452.25799368793747,"max_cost":723.6127899006999},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-max","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.474999999999999e-07,"completion":3.7374999999999994e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4949999999999998e-07,"input_cache_write":9.343749999999998e-07,"max_prompt_cost":0.7474999999999999,"max_completion_cost":0.24494079999999996,"max_cost":0.94345264},"sats_pricing":{"prompt":0.0011501515545856158,"completion":0.005750757772928079,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023003031091712315,"input_cache_write":0.0014376894432320197,"max_prompt_cost":1150.1515545856157,"max_completion_cost":376.88166140661457,"max_cost":1451.6568837109076},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-coder-plus","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-codex:batch","name":"OpenAI: GPT-5 Codex (batch)","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.1875e-07,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":7.187499999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.2875,"max_completion_cost":0.736,"max_cost":0.9315},"sats_pricing":{"prompt":0.001105914956332323,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00011059149563323228,"input_cache_write":0.0,"max_prompt_cost":442.36598253292914,"max_completion_cost":1132.4569152842987,"max_cost":1433.2657834066906},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-codex","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":3.105e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5525e-07,"input_cache_write":0.0,"max_prompt_cost":0.040697856,"max_completion_cost":0.03768319999999999,"max_cost":0.06820659199999998},"sats_pricing":{"prompt":0.0004777552611355635,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023887763056778174,"input_cache_write":0.0,"max_prompt_cost":62.62033758756058,"max_completion_cost":57.981794062556084,"max_cost":104.9470472532265},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-v3.1-terminus","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2424999999999999e-07,"completion":1.1212499999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4849999999999995e-08,"input_cache_write":2.8031249999999996e-07,"max_prompt_cost":0.22424999999999998,"max_completion_cost":0.07348223999999999,"max_cost":0.28303579199999995},"sats_pricing":{"prompt":0.00034504546637568475,"completion":0.0017252273318784236,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.900909327513695e-05,"input_cache_write":0.0004313068329696059,"max_prompt_cost":345.0454663756847,"max_completion_cost":113.06449842198437,"max_cost":435.4970651132722},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-coder-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.36175872,"max_cost":0.36175872},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":556.6252230005385,"max_cost":556.6252230005385},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.035e-07,"completion":1.265e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027131904,"max_completion_cost":0.02072576,"max_cost":0.046161919999999995},"sats_pricing":{"prompt":0.0001592517537118545,"completion":0.0019464103231448884,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":41.746891725040385,"max_completion_cost":31.889986734405852,"max_cost":71.0276977266312},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.9899999999999996e-07,"completion":8.969999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.299,"max_completion_cost":0.029392895999999998,"max_cost":0.318595264},"sats_pricing":{"prompt":0.0004600606218342463,"completion":0.001380181865502739,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":460.0606218342463,"max_completion_cost":45.22579936879375,"max_cost":490.2111547467755},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":5.749999999999999e-07,"max_prompt_cost":0.45999999999999996,"max_completion_cost":0.04521984,"max_cost":0.49014655999999995},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0008847319650658582,"max_prompt_cost":707.7855720526867,"max_completion_cost":69.57815287506732,"max_cost":754.1710073027315},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":0.288512,"max_cost":0.40014848},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":443.9231107914451,"max_cost":615.6941757017674},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2-0905","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":2.76e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018841599999999997,"max_completion_cost":0.09043968,"max_cost":0.10174464},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0042467134323161205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.990897031278042,"max_completion_cost":139.15630575013463,"max_cost":156.55084396890146},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-70b","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":1.4949999999999998e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019595263999999998,"max_completion_cost":0.06029311999999999,"max_cost":0.06029311999999999},"sats_pricing":{"prompt":0.00023003031091712315,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.150532912529165,"max_completion_cost":92.77087050008974,"max_cost":92.77087050008974},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nousresearch/hermes-4-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-405b","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":3.45e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.4521984,"max_cost":0.4521984},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.00530839179039515,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":231.92717625022433,"max_completion_cost":695.7815287506731,"max_cost":695.7815287506731},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nousresearch/hermes-4-405b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.0925e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4949999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.04710399999999999,"max_completion_cost":0.03579904,"max_cost":0.07348223999999999},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0016809907336251307,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023003031091712315,"input_cache_write":0.0,"max_prompt_cost":72.4772425781951,"max_completion_cost":55.082704359428284,"max_cost":113.06449842198437},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-chat-v3.1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5999999999999995e-08,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.30146559999999994,"max_cost":0.30146559999999994},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.077855720526867e-05,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":463.85435250044867,"max_cost":463.85435250044867},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-medium-3.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5v","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":2.0699999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.265e-07,"input_cache_write":0.0,"max_prompt_cost":0.04521984,"max_completion_cost":0.033914879999999994,"max_cost":0.06782975999999999},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.0031850350742370897,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00019464103231448885,"input_cache_write":0.0,"max_prompt_cost":69.57815287506732,"max_completion_cost":52.18361465630048,"max_cost":104.36722931260095},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.5v","alias_ids":null,"forwarded_model_id":null},{"id":"jamba-large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":9.199999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5887999999999999,"max_completion_cost":0.03768319999999999,"max_cost":0.6170623999999999},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.014155711441053731,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":905.9655322274388,"max_completion_cost":57.981794062556084,"max_cost":949.4518777743559},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"ai21/jamba-large-1.7","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.575,"max_completion_cost":1.472,"max_cost":1.863},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0,"max_prompt_cost":884.7319650658583,"max_completion_cost":2264.9138305685974,"max_cost":2866.5315668133812},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5:batch","name":"OpenAI: GPT-5 (batch)","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.1875e-07,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":7.187499999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.2875,"max_completion_cost":0.736,"max_cost":0.9315},"sats_pricing":{"prompt":0.001105914956332323,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00011059149563323228,"input_cache_write":0.0,"max_prompt_cost":442.36598253292914,"max_completion_cost":1132.4569152842987,"max_cost":1433.2657834066906},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.11499999999999998,"max_completion_cost":0.29439999999999994,"max_cost":0.37259999999999993},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":176.94639301317164,"max_completion_cost":452.9827661137194,"max_cost":573.3063133626762},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini:batch","name":"OpenAI: GPT-5 Mini (batch)","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4374999999999997e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.05749999999999999,"max_completion_cost":0.14719999999999997,"max_cost":0.18629999999999997},"sats_pricing":{"prompt":0.00022118299126646455,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":2.2118299126646457e-05,"input_cache_write":0.0,"max_prompt_cost":88.47319650658582,"max_completion_cost":226.4913830568597,"max_cost":286.6531566813381},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-09,"input_cache_write":0.0,"max_prompt_cost":0.022999999999999996,"max_completion_cost":0.058879999999999995,"max_cost":0.07451999999999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-06,"input_cache_write":0.0,"max_prompt_cost":35.38927860263433,"max_completion_cost":90.59655322274389,"max_cost":114.66126267253523},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano:batch","name":"OpenAI: GPT-5 Nano (batch)","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8749999999999996e-08,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999997e-09,"input_cache_write":0.0,"max_prompt_cost":0.011499999999999998,"max_completion_cost":0.029439999999999997,"max_cost":0.037259999999999995},"sats_pricing":{"prompt":4.4236598253292915e-05,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":4.423659825329292e-06,"input_cache_write":0.0,"max_prompt_cost":17.694639301317164,"max_completion_cost":45.298276611371946,"max_cost":57.330631336267615},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.255e-08,"completion":1.9549999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0055771136,"max_completion_cost":0.025624575999999996,"max_cost":0.025624575999999996},"sats_pricing":{"prompt":6.547016541487353e-05,"completion":0.0003008088681223918,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.581305521258303,"max_completion_cost":39.42761996253814,"max_cost":39.42761996253814},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-oss-120b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.449999999999999e-08,"completion":1.4949999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.004521983999999999,"max_completion_cost":0.019595263999999998,"max_cost":0.019595263999999998},"sats_pricing":{"prompt":5.308391790395149e-05,"completion":0.00023003031091712315,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":6.95781528750673,"max_completion_cost":30.150532912529165,"max_cost":30.150532912529165},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-oss-20b","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":8.624999999999998e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.725e-06,"input_cache_write":2.1562499999999996e-05,"max_prompt_cost":3.4499999999999997,"max_completion_cost":2.7599999999999993,"max_cost":5.6579999999999995},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.13270979475987874,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.002654195895197575,"input_cache_write":0.033177448689969684,"max_prompt_cost":5308.39179039515,"max_completion_cost":4246.713432316119,"max_cost":8705.762536248047},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1:batch","name":"Anthropic: Claude Opus 4.1 (batch)","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":8.625e-06,"completion":4.312499999999999e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":8.625e-07,"input_cache_write":1.0781249999999998e-05,"max_prompt_cost":1.7249999999999999,"max_completion_cost":1.3799999999999997,"max_cost":2.8289999999999997},"sats_pricing":{"prompt":0.013270979475987875,"completion":0.06635489737993937,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0013270979475987876,"input_cache_write":0.016588724344984842,"max_prompt_cost":2654.195895197575,"max_completion_cost":2123.3567161580595,"max_cost":4352.881268124023},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-2508","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.0349999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.08832,"max_completion_cost":0.26496,"max_cost":0.26496},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0015925175371185448,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":135.89482983411585,"max_completion_cost":407.6844895023475,"max_cost":407.6844895023475},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/codestral-2508","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.05e-08,"completion":3.105e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012879999999999999,"max_completion_cost":0.010174464,"max_cost":0.02041664},"sats_pricing":{"prompt":0.00012386247510922017,"completion":0.0004777552611355635,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.817996017475227,"max_completion_cost":15.655084396890144,"max_cost":31.414354829986447},"per_request_limits":null,"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":5.5372499999999995e-08,"completion":2.2200749999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007087679999999999,"max_completion_cost":0.0071042399999999995,"max_cost":0.012419999999999999},"sats_pricing":{"prompt":8.519968823584216e-05,"completion":0.0003415950117119279,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.905560094187795,"max_completion_cost":10.931040374781693,"max_cost":19.11021044542254},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.9e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.265e-07,"input_cache_write":0.0,"max_prompt_cost":0.09043968,"max_completion_cost":0.24870912,"max_cost":0.27131904},"sats_pricing":{"prompt":0.0010616783580790301,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00019464103231448885,"input_cache_write":0.0,"max_prompt_cost":139.15630575013463,"max_completion_cost":382.6798408128702,"max_cost":417.46891725040393},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.5","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5-air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4949999999999998e-07,"completion":9.775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.019595263999999998,"max_completion_cost":0.09609216,"max_cost":0.100990976},"sats_pricing":{"prompt":0.00023003031091712315,"completion":0.0015040443406119592,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":30.150532912529165,"max_completion_cost":147.85357485951803,"max_cost":155.39120808765034},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"z-ai/glm-4.5-air","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.6449999999999997e-07,"completion":2.6449999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.034668543999999996,"max_completion_cost":0.34668543999999996,"max_cost":0.34668543999999996},"sats_pricing":{"prompt":0.00040697670393029483,"completion":0.004069767039302948,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53.343250537551604,"max_completion_cost":533.432505375516,"max_cost":533.432505375516},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.09043968,"max_completion_cost":0.07536639999999999,"max_cost":0.14319615999999996},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":139.15630575013463,"max_completion_cost":115.96358812511217,"max_cost":220.3308174377131},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.014719999999999999,"max_completion_cost":0.00047103999999999994,"max_cost":0.014955519999999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":22.649138305685973,"max_completion_cost":0.7247724257819511,"max_cost":23.011524518576948},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"bytedance/ui-tars-1.5-7b","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":1.1499999999999998e-07,"web_search":0.0161,"internal_reasoning":4.5999999999999994e-07,"input_cache_read":1.1499999999999999e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.12058623999999998,"max_completion_cost":0.030146099999999995,"max_cost":0.14319581499999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0007077855720526866,"request":0.001,"image":0.00017694639301317166,"web_search":24.772495021844033,"internal_reasoning":0.0007077855720526866,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":185.54174100017948,"max_completion_cost":46.38472746447282,"max_cost":220.3302865985341},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite:batch","name":"Google: Gemini 2.5 Flash Lite (batch)","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":2.2999999999999997e-07,"request":0.0,"image":5.749999999999999e-08,"web_search":0.0161,"internal_reasoning":2.2999999999999997e-07,"input_cache_read":1.1499999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.015073049999999998,"max_cost":0.07159790749999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.0003538927860263433,"request":0.001,"image":8.847319650658583e-05,"web_search":24.772495021844033,"internal_reasoning":0.0003538927860263433,"input_cache_read":1.7694639301317167e-05,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":23.19236373223641,"max_cost":110.16514329926704},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.035e-07,"completion":6.325e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027131904,"max_completion_cost":0.01036288,"max_cost":0.03579904},"sats_pricing":{"prompt":0.0001592517537118545,"completion":0.0009732051615724442,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":41.746891725040385,"max_completion_cost":15.944993367202926,"max_cost":55.082704359428284},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-235b-a22b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.555e-07,"completion":2.6449999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.085917696,"max_completion_cost":0.26543103999999995,"max_cost":0.28556799999999993},"sats_pricing":{"prompt":0.0010085944401750785,"completion":0.004069767039302948,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":132.1984904626279,"max_completion_cost":408.4092619281294,"max_cost":439.3932831303078},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"moonshotai/kimi-k2","alias_ids":null,"forwarded_model_id":null},{"id":"dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":1.0349999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.029439999999999997,"max_completion_cost":0.008478719999999999,"max_cost":0.03603455999999999},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0015925175371185448,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":45.298276611371946,"max_completion_cost":13.04590366407512,"max_cost":55.44509057231926},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"venice/uncensored","alias_ids":null,"forwarded_model_id":null},{"id":"hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.61e-07,"completion":6.555e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021102592,"max_completion_cost":0.085917696,"max_cost":0.085917696},"sats_pricing":{"prompt":0.00024772495021844034,"completion":0.0010085944401750785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.46980467503141,"max_completion_cost":132.1984904626279,"max_cost":132.1984904626279},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"tencent/hunyuan-a13b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-large","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0349999999999998e-06,"completion":2.185e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27131903999999996,"max_completion_cost":0.28639232,"max_cost":0.42205183999999996},"sats_pricing":{"prompt":0.0015925175371185448,"completion":0.0033619814672502615,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":417.4689172504038,"max_completion_cost":440.6616348754263,"max_cost":649.3960935006282},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"morph/morph-v3-large","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-fast","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.199999999999999e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07536639999999999,"max_completion_cost":0.052439999999999994,"max_cost":0.0928464},"sats_pricing":{"prompt":0.0014155711441053733,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":115.96358812511217,"max_completion_cost":80.68755521400628,"max_cost":142.8594398631143},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"morph/morph-v3-fast","alias_ids":null,"forwarded_model_id":null},{"id":"ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.83e-07,"completion":1.4375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.059409,"max_completion_cost":0.023,"max_cost":0.074681},"sats_pricing":{"prompt":0.000743174850655321,"completion":0.002211829912664646,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":91.41050663060449,"max_completion_cost":35.389278602634334,"max_cost":114.90898762275368},"per_request_limits":null,"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.078125e-07,"completion":2.8749999999999995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0276,"max_completion_cost":0.004710399999999999,"max_cost":0.030544},"sats_pricing":{"prompt":0.00016588724344984845,"completion":0.0004423659825329291,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.467134323161204,"max_completion_cost":7.2477242578195105,"max_cost":46.9969619842984},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m1","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6325,"max_completion_cost":0.1012,"max_cost":0.7083999999999999},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":973.2051615724441,"max_completion_cost":155.71282585159108,"max_cost":1089.9897809611375},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-m1","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.45e-07,"completion":2.875e-06,"request":0.0,"image":3.45e-07,"web_search":0.0161,"internal_reasoning":2.875e-06,"input_cache_read":3.449999999999999e-08,"input_cache_write":9.583333333333328e-08,"max_prompt_cost":0.36175872,"max_completion_cost":0.188413125,"max_cost":0.52756227},"sats_pricing":{"prompt":0.0005308391790395151,"completion":0.004423659825329292,"request":0.001,"image":0.0005308391790395151,"web_search":24.772495021844033,"internal_reasoning":0.004423659825329292,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.00014745532751097632,"max_prompt_cost":556.6252230005385,"max_completion_cost":289.9045466529551,"max_cost":811.741224055139},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash:batch","name":"Google: Gemini 2.5 Flash (batch)","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":1.4375e-06,"request":0.0,"image":1.725e-07,"web_search":0.0161,"internal_reasoning":1.4375e-06,"input_cache_read":3.449999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.18087936,"max_completion_cost":0.0942065625,"max_cost":0.263781135},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.002211829912664646,"request":0.001,"image":0.00026541958951975753,"web_search":24.772495021844033,"internal_reasoning":0.002211829912664646,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0,"max_prompt_cost":278.31261150026927,"max_completion_cost":144.95227332647755,"max_cost":405.8706120275695},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":1.4375e-06,"web_search":0.0161,"internal_reasoning":1.15e-05,"input_cache_read":1.4374999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":1.507328,"max_completion_cost":0.753664,"max_cost":2.166784},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.002211829912664646,"web_search":24.772495021844033,"internal_reasoning":0.017694639301317167,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0006635489737993938,"max_prompt_cost":2319.2717625022437,"max_completion_cost":1159.6358812511219,"max_cost":3333.9531585969753},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro:batch","name":"Google: Gemini 2.5 Pro (batch)","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.1875e-07,"completion":5.75e-06,"request":0.0,"image":7.1875e-07,"web_search":0.0161,"internal_reasoning":5.75e-06,"input_cache_read":1.4374999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.753664,"max_completion_cost":0.376832,"max_cost":1.083392},"sats_pricing":{"prompt":0.001105914956332323,"completion":0.008847319650658584,"request":0.001,"image":0.001105914956332323,"web_search":24.772495021844033,"internal_reasoning":0.008847319650658584,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0,"max_prompt_cost":1159.6358812511219,"max_completion_cost":579.8179406255609,"max_cost":1666.9765792984876},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.3e-05,"completion":9.2e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.6,"max_completion_cost":9.2,"max_cost":11.5},"sats_pricing":{"prompt":0.035389278602634335,"completion":0.14155711441053734,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7077.855720526866,"max_completion_cost":14155.711441053732,"max_cost":17694.63930131717},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro:batch","name":"OpenAI: o3 Pro (batch)","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":4.6e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.3,"max_completion_cost":4.6,"max_cost":5.75},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.07077855720526867,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3538.927860263433,"max_completion_cost":7077.855720526866,"max_cost":8847.319650658585},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":1.4375e-06,"web_search":0.0161,"internal_reasoning":1.15e-05,"input_cache_read":1.4374999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":1.507328,"max_completion_cost":0.753664,"max_cost":2.166784},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.002211829912664646,"web_search":24.772495021844033,"internal_reasoning":0.017694639301317167,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0006635489737993938,"max_prompt_cost":2319.2717625022437,"max_completion_cost":1159.6358812511219,"max_cost":3333.9531585969753},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-pro-preview-06-05","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":5.749999999999999e-07,"completion":2.4725e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.0249999999999996e-07,"input_cache_write":0.0,"max_prompt_cost":0.09420799999999999,"max_completion_cost":0.08101888,"max_cost":0.15638528},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.003804347449783191,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0006193123755461008,"input_cache_write":0.0,"max_prompt_cost":144.9544851563902,"max_completion_cost":124.66085723449561,"max_cost":240.6244453596078},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-r1-0528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":8.624999999999998e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.725e-06,"input_cache_write":2.1562499999999996e-05,"max_prompt_cost":3.4499999999999997,"max_completion_cost":2.7599999999999993,"max_cost":5.6579999999999995},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.13270979475987874,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.002654195895197575,"input_cache_write":0.033177448689969684,"max_prompt_cost":5308.39179039515,"max_completion_cost":4246.713432316119,"max_cost":8705.762536248047},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4-opus-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.45e-07,"input_cache_write":4.3125e-06,"max_prompt_cost":0.69,"max_completion_cost":1.1039999999999999,"max_cost":1.5732},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0005308391790395151,"input_cache_write":0.006635489737993937,"max_prompt_cost":1061.67835807903,"max_completion_cost":1698.6853729264478,"max_cost":2420.6266564201883},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-4-sonnet-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.899999999999998e-08,"completion":1.3799999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0022609919999999994,"max_completion_cost":0.004521983999999999,"max_cost":0.004521983999999999},"sats_pricing":{"prompt":0.00010616783580790298,"completion":0.00021233567161580596,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.478907643753365,"max_completion_cost":6.95781528750673,"max_cost":6.95781528750673},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-3n-e4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5999999999999995e-08,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.30146559999999994,"max_cost":0.30146559999999994},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.077855720526867e-05,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":463.85435250044867,"max_cost":463.85435250044867},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-medium-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":1.15e-05,"request":0.0,"image":1.4375e-06,"web_search":0.0161,"internal_reasoning":1.15e-05,"input_cache_read":1.4374999999999997e-07,"input_cache_write":4.3125e-07,"max_prompt_cost":1.507328,"max_completion_cost":0.7536525,"max_cost":2.1667739375},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.017694639301317167,"request":0.001,"image":0.002211829912664646,"web_search":24.772495021844033,"internal_reasoning":0.017694639301317167,"input_cache_read":0.00022118299126646455,"input_cache_write":0.0006635489737993938,"max_prompt_cost":2319.2717625022437,"max_completion_cost":1159.6181866118204,"max_cost":3333.9376757875866},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemini-2.5-pro-preview-03-25","alias_ids":null,"forwarded_model_id":null},{"id":"virtuoso-large","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.625e-07,"completion":1.38e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1130496,"max_completion_cost":0.08832,"max_cost":0.1461696},"sats_pricing":{"prompt":0.0013270979475987876,"completion":0.0021233567161580602,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":173.9453821876683,"max_completion_cost":135.89482983411585,"max_cost":224.90594337546176},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"arcee-ai/virtuoso-large","alias_ids":null,"forwarded_model_id":null},{"id":"llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.07e-07,"completion":2.07e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.033914879999999994,"max_completion_cost":0.003391488,"max_cost":0.033914879999999994},"sats_pricing":{"prompt":0.000318503507423709,"completion":0.000318503507423709,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":52.18361465630048,"max_completion_cost":5.218361465630048,"max_cost":52.18361465630048},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-guard-4-12b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.3799999999999997e-07,"completion":5.749999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005652479999999999,"max_completion_cost":0.009420799999999998,"max_cost":0.012812287999999998},"sats_pricing":{"prompt":0.00021233567161580596,"completion":0.0008847319650658582,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.697269109383413,"max_completion_cost":14.495448515639021,"max_cost":19.71380998126907},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-30b-a3b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-8b","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.3455e-07,"completion":5.232499999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0176357376,"max_completion_cost":0.004286463999999999,"max_cost":0.020819968},"sats_pricing":{"prompt":0.00020702727982541087,"completion":0.000805106088209931,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.135479621276254,"max_completion_cost":6.595429074615755,"max_cost":32.03494121956224},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-8b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-14b","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.6162499999999996e-07,"completion":1.0464999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.034291711999999995,"max_completion_cost":0.008572927999999999,"max_cost":0.040721407999999994},"sats_pricing":{"prompt":0.0004025530441049655,"completion":0.001610212176419862,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":52.76343259692604,"max_completion_cost":13.19085814923151,"max_cost":62.65657620884967},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-14b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-32b","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":9.199999999999999e-08,"completion":3.22e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0037683199999999995,"max_completion_cost":0.005275648,"max_cost":0.00753664},"sats_pricing":{"prompt":0.00014155711441053733,"completion":0.0004954499004368807,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.798179406255609,"max_completion_cost":8.117451168757853,"max_cost":11.59635881251122},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-32b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":5.232499999999999e-07,"completion":2.0929999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06858342399999999,"max_completion_cost":0.017145855999999998,"max_cost":0.08144281599999999},"sats_pricing":{"prompt":0.000805106088209931,"completion":0.003220424352839724,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":105.52686519385207,"max_completion_cost":26.38171629846302,"max_cost":125.31315241769934},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen3-235b-a22b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.265e-06,"completion":5.06e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.1625e-07,"input_cache_write":0.0,"max_prompt_cost":0.253,"max_completion_cost":0.506,"max_cost":0.6325000000000001},"sats_pricing":{"prompt":0.0019464103231448884,"completion":0.007785641292579554,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004866025807862221,"input_cache_write":0.0,"max_prompt_cost":389.2820646289777,"max_completion_cost":778.5641292579554,"max_cost":973.2051615724444},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high:batch","name":"OpenAI: o4 Mini High (batch)","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.58125e-07,"input_cache_write":0.0,"max_prompt_cost":0.1265,"max_completion_cost":0.253,"max_cost":0.31625000000000003},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00024330129039311105,"input_cache_write":0.0,"max_prompt_cost":194.64103231448885,"max_completion_cost":389.2820646289777,"max_cost":486.6025807862222},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":9.199999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":0.4599999999999999,"max_completion_cost":0.9199999999999998,"max_cost":1.1499999999999997},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.014155711441053731,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.0,"max_prompt_cost":707.7855720526866,"max_completion_cost":1415.5711441053732,"max_cost":1769.4639301317163},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3:batch","name":"OpenAI: o3 (batch)","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":0.22999999999999995,"max_completion_cost":0.4599999999999999,"max_cost":0.5749999999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":353.8927860263433,"max_completion_cost":707.7855720526866,"max_cost":884.7319650658582},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.265e-06,"completion":5.06e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.1625e-07,"input_cache_write":0.0,"max_prompt_cost":0.253,"max_completion_cost":0.506,"max_cost":0.6325000000000001},"sats_pricing":{"prompt":0.0019464103231448884,"completion":0.007785641292579554,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004866025807862221,"input_cache_write":0.0,"max_prompt_cost":389.2820646289777,"max_completion_cost":778.5641292579554,"max_cost":973.2051615724444},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini:batch","name":"OpenAI: o4 Mini (batch)","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.58125e-07,"input_cache_write":0.0,"max_prompt_cost":0.1265,"max_completion_cost":0.253,"max_cost":0.31625000000000003},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00024330129039311105,"input_cache_write":0.0,"max_prompt_cost":194.64103231448885,"max_completion_cost":389.2820646289777,"max_cost":486.6025807862222},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":9.199999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":2.4094247999999996,"max_completion_cost":0.30146559999999994,"max_cost":2.6355239999999993},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.014155711441053731,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0008847319650658582,"input_cache_write":0.0,"max_prompt_cost":3707.295892143326,"max_completion_cost":463.85435250044867,"max_cost":4055.1866565186624},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1:batch","name":"OpenAI: GPT-4.1 (batch)","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":1.2047123999999998,"max_completion_cost":0.15073279999999997,"max_cost":1.3177619999999997},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":1853.647946071663,"max_completion_cost":231.92717625022433,"max_cost":2027.5933282593312},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.5999999999999994e-07,"completion":1.8399999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.1499999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.4818849599999999,"max_completion_cost":0.06029311999999999,"max_cost":0.5271047999999999},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.0028311422882107465,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.00017694639301317166,"input_cache_write":0.0,"max_prompt_cost":741.4591784286652,"max_completion_cost":92.77087050008974,"max_cost":811.0373313037326},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini:batch","name":"OpenAI: GPT-4.1 Mini (batch)","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":9.199999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":5.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.24094247999999996,"max_completion_cost":0.030146559999999996,"max_cost":0.26355239999999996},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0014155711441053733,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":8.847319650658583e-05,"input_cache_write":0.0,"max_prompt_cost":370.7295892143326,"max_completion_cost":46.38543525004487,"max_cost":405.5186656518663},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.12047123999999998,"max_completion_cost":0.015073279999999998,"max_cost":0.13177619999999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":185.3647946071663,"max_completion_cost":23.192717625022436,"max_cost":202.75933282593314},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano:batch","name":"OpenAI: GPT-4.1 Nano (batch)","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":1.4374999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.06023561999999999,"max_completion_cost":0.007536639999999999,"max_cost":0.06588809999999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":2.2118299126646457e-05,"input_cache_write":0.0,"max_prompt_cost":92.68239730358314,"max_completion_cost":11.596358812511218,"max_cost":101.37966641296657},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-maverick","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":9.199999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.24117247999999997,"max_completion_cost":0.015073279999999998,"max_cost":0.25247744},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":371.08348200035897,"max_completion_cost":23.192717625022436,"max_cost":388.4780202191258},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-scout","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.03768319999999999,"max_completion_cost":0.00565248,"max_cost":0.04145152},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":57.981794062556084,"max_completion_cost":8.697269109383415,"max_cost":63.779973468811704},"per_request_limits":null,"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":3.105e-07,"completion":1.288e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5525e-07,"input_cache_write":0.0,"max_prompt_cost":0.05087232,"max_completion_cost":0.084410368,"max_cost":0.11493376},"sats_pricing":{"prompt":0.0004777552611355635,"completion":0.0019817996017475227,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023887763056778174,"input_cache_write":0.0,"max_prompt_cost":78.27542198445073,"max_completion_cost":129.87921870012565,"max_cost":176.8444718907961},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-chat-v3-0324","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":0.00017249999999999996,"completion":0.0006899999999999999,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.49999999999999,"max_completion_cost":68.99999999999999,"max_cost":86.24999999999999},"sats_pricing":{"prompt":0.26541958951975747,"completion":1.0616783580790299,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53083.91790395149,"max_completion_cost":106167.83580790299,"max_cost":132709.79475987874},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro:batch","name":"OpenAI: o1-pro (batch)","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.624999999999998e-05,"completion":0.00034499999999999993,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17.249999999999996,"max_completion_cost":34.49999999999999,"max_cost":43.12499999999999},"sats_pricing":{"prompt":0.13270979475987874,"completion":0.5308391790395149,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":26541.958951975746,"max_completion_cost":53083.91790395149,"max_cost":66354.89737993937},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.0365e-07,"completion":6.382499999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051667199999999996,"max_completion_cost":0.08169599999999999,"max_cost":0.08169599999999999},"sats_pricing":{"prompt":0.0006210818394762326,"completion":0.0009820524812231026,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":79.49847545295776,"max_completion_cost":125.70271759655715,"max_cost":125.70271759655715},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.749999999999999e-08,"completion":1.1499999999999998e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.0018841599999999997,"max_cost":0.008478719999999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.00017694639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":2.8990897031278045,"max_cost":13.04590366407512},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-3-4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.749999999999999e-08,"completion":1.725e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.00282624,"max_cost":0.0094208},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.00026541958951975753,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":4.348634554691707,"max_cost":14.495448515639024},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-3-12b-it","alias_ids":null,"forwarded_model_id":null},{"id":"command-a","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.736,"max_completion_cost":0.094208,"max_cost":0.8066559999999999},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1132.4569152842987,"max_completion_cost":144.95448515639023,"max_cost":1241.1727791515914},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"cohere/command-a-03-2025","alias_ids":null,"forwarded_model_id":null},{"id":"reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-07,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.015073279999999998,"max_cost":0.015073279999999998},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":23.192717625022436,"max_cost":23.192717625022436},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"rekaai/reka-flash-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":9.199999999999999e-08,"completion":5.174999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5999999999999995e-08,"input_cache_write":0.0,"max_prompt_cost":0.012058623999999999,"max_completion_cost":0.06782975999999999,"max_cost":0.06782975999999999},"sats_pricing":{"prompt":0.00014155711441053733,"completion":0.0007962587685592724,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.077855720526867e-05,"input_cache_write":0.0,"max_prompt_cost":18.55417410001795,"max_completion_cost":104.36722931260095,"max_cost":104.36722931260095},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-3-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":9.199999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999995e-07,"input_cache_write":0.0,"max_prompt_cost":0.02072576,"max_completion_cost":0.030146559999999996,"max_cost":0.030146559999999996},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.0014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004423659825329291,"input_cache_write":0.0,"max_prompt_cost":31.889986734405852,"max_completion_cost":46.38543525004487,"max_cost":46.38543525004487},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thedrummer/skyfall-36b-v2","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.2999999999999996e-06,"completion":9.199999999999998e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29439999999999994,"max_completion_cost":1.1775999999999998,"max_cost":1.1775999999999998},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.014155711441053731,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":452.9827661137194,"max_completion_cost":1811.9310644548775,"max_cost":1811.9310644548775},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perplexity/sonar-reasoning-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.69,"max_completion_cost":0.13799999999999998,"max_cost":0.8004},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1061.67835807903,"max_completion_cost":212.33567161580598,"max_cost":1231.546895371675},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perplexity/sonar-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-deep-research","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.2999999999999996e-06,"completion":9.199999999999998e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":3.45e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29439999999999994,"max_completion_cost":1.1775999999999998,"max_cost":1.1775999999999998},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.014155711441053731,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.00530839179039515,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":452.9827661137194,"max_completion_cost":1811.9310644548775,"max_cost":1811.9310644548775},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perplexity/sonar-deep-research","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-saba","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.02260992,"max_cost":0.02260992},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.538927860263433e-05,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":34.78907643753366,"max_cost":34.78907643753366},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-saba-2502","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.265e-06,"completion":5.06e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":6.325e-07,"input_cache_write":0.0,"max_prompt_cost":0.253,"max_completion_cost":0.506,"max_cost":0.6325000000000001},"sats_pricing":{"prompt":0.0019464103231448884,"completion":0.007785641292579554,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0009732051615724442,"input_cache_write":0.0,"max_prompt_cost":389.2820646289777,"max_completion_cost":778.5641292579554,"max_cost":973.2051615724444},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high:batch","name":"OpenAI: o3 Mini High (batch)","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.1625e-07,"input_cache_write":0.0,"max_prompt_cost":0.1265,"max_completion_cost":0.253,"max_cost":0.31625000000000003},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004866025807862221,"input_cache_write":0.0,"max_prompt_cost":194.64103231448885,"max_completion_cost":389.2820646289777,"max_cost":486.6025807862222},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.199999999999999e-07,"completion":1.8399999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030146559999999996,"max_completion_cost":0.06029311999999999,"max_cost":0.06029311999999999},"sats_pricing":{"prompt":0.0014155711441053733,"completion":0.0028311422882107465,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.38543525004487,"max_completion_cost":92.77087050008974,"max_cost":92.77087050008974},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":8.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.009199999999999998,"max_completion_cost":0.0276,"max_cost":0.0276},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0013270979475987876,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14.15571144105373,"max_completion_cost":42.467134323161204,"max_cost":42.467134323161204},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.9899999999999996e-07,"completion":8.969999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.979999999999999e-08,"input_cache_write":3.7374999999999997e-07,"max_prompt_cost":0.299,"max_completion_cost":0.029392895999999998,"max_cost":0.318595264},"sats_pricing":{"prompt":0.0004600606218342463,"completion":0.001380181865502739,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.201212436684926e-05,"input_cache_write":0.0005750757772928079,"max_prompt_cost":460.0606218342463,"max_completion_cost":45.22579936879375,"max_cost":490.2111547467755},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-plus-2025-01-25","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.265e-06,"completion":5.06e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":6.325e-07,"input_cache_write":0.0,"max_prompt_cost":0.253,"max_completion_cost":0.506,"max_cost":0.6325000000000001},"sats_pricing":{"prompt":0.0019464103231448884,"completion":0.007785641292579554,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0009732051615724442,"input_cache_write":0.0,"max_prompt_cost":389.2820646289777,"max_completion_cost":778.5641292579554,"max_cost":973.2051615724444},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini:batch","name":"OpenAI: o3 Mini (batch)","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.325e-07,"completion":2.53e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.1625e-07,"input_cache_write":0.0,"max_prompt_cost":0.1265,"max_completion_cost":0.253,"max_cost":0.31625000000000003},"sats_pricing":{"prompt":0.0009732051615724442,"completion":0.003892820646289777,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0004866025807862221,"input_cache_write":0.0,"max_prompt_cost":194.64103231448885,"max_completion_cost":389.2820646289777,"max_cost":486.6025807862222},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.749999999999999e-08,"completion":9.199999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0018841599999999997,"max_completion_cost":0.0015073279999999998,"max_cost":0.0024494079999999997},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.00014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8990897031278045,"max_completion_cost":2.3192717625022437,"max_cost":3.7688166140661457},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","alias_ids":null,"forwarded_model_id":null},{"id":"sonar","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.00575,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14613279999999998,"max_completion_cost":0.14613279999999998,"max_cost":0.14613279999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":8.847319650658584,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":224.84932052969748,"max_completion_cost":224.84932052969748,"max_cost":224.84932052969748},"per_request_limits":null,"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"perplexity/sonar","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":9.199999999999999e-07,"completion":9.199999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.007536639999999999,"max_cost":0.007536639999999999},"sats_pricing":{"prompt":0.0014155711441053733,"completion":0.0014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":11.596358812511218,"max_cost":11.596358812511218},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":8.049999999999999e-07,"completion":2.875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051519999999999996,"max_completion_cost":0.046,"max_cost":0.08463999999999999},"sats_pricing":{"prompt":0.0012386247510922017,"completion":0.004423659825329292,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":79.27198406990091,"max_completion_cost":70.77855720526867,"max_cost":130.23254525769434},"per_request_limits":null,"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-r1","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2999999999999997e-07,"completion":1.265e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.23004415999999997,"max_completion_cost":1.26524288,"max_cost":1.26524288},"sats_pricing":{"prompt":0.0003538927860263433,"completion":0.0019464103231448884,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":353.96073344126034,"max_completion_cost":1946.7840339269321,"max_cost":1946.7840339269321},"per_request_limits":null,"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"minimax/minimax-01","alias_ids":null,"forwarded_model_id":null},{"id":"phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.05e-08,"completion":1.61e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001318912,"max_completion_cost":0.002637824,"max_cost":0.002637824},"sats_pricing":{"prompt":0.00012386247510922017,"completion":0.00024772495021844034,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.0293627921894632,"max_completion_cost":4.0587255843789265,"max_cost":4.0587255843789265},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"microsoft/phi-4","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.9601e-07,"completion":1.1830049999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.037889280000000004,"max_completion_cost":0.018928079999999996,"max_cost":0.052081199999999994},"sats_pricing":{"prompt":0.00045546001561590395,"completion":0.0018202475449264968,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":58.298881998835704,"max_completion_cost":29.123960718823948,"max_cost":80.13548246780518},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"deepseek/deepseek-chat-v3","alias_ids":null,"forwarded_model_id":null},{"id":"l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":7.474999999999999e-07,"completion":8.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09797631999999999,"max_completion_cost":0.0141312,"max_cost":0.09986047999999999},"sats_pricing":{"prompt":0.0011501515545856158,"completion":0.0013270979475987876,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":150.75266456264583,"max_completion_cost":21.743172773458536,"max_cost":153.65175426577363},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","alias_ids":null,"forwarded_model_id":null},{"id":"o1","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-05,"completion":6.9e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":8.625e-06,"input_cache_write":0.0,"max_prompt_cost":3.4499999999999997,"max_completion_cost":6.8999999999999995,"max_cost":8.625},"sats_pricing":{"prompt":0.02654195895197575,"completion":0.106167835807903,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.013270979475987875,"input_cache_write":0.0,"max_prompt_cost":5308.39179039515,"max_completion_cost":10616.7835807903,"max_cost":13270.979475987875},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"o1:batch","name":"OpenAI: o1 (batch)","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.625e-06,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":4.3125e-06,"input_cache_write":0.0,"max_prompt_cost":1.7249999999999999,"max_completion_cost":3.4499999999999997,"max_cost":4.3125},"sats_pricing":{"prompt":0.013270979475987875,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.006635489737993937,"input_cache_write":0.0,"max_prompt_cost":2654.195895197575,"max_completion_cost":5308.39179039515,"max_cost":6635.4897379939375},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":4.3125e-08,"completion":1.725e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00552,"max_completion_cost":0.00069,"max_cost":0.0060374999999999995},"sats_pricing":{"prompt":6.635489737993938e-05,"completion":0.00026541958951975753,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.49342686463224,"max_completion_cost":1.06167835807903,"max_cost":9.289685633191512},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"cohere/command-r7b-12-2024","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":1.1499999999999998e-07,"completion":3.6799999999999996e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.015073279999999998,"max_completion_cost":0.006029311999999999,"max_cost":0.019218431999999997},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0005662284576421493,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":23.192717625022436,"max_completion_cost":9.277087050008975,"max_cost":29.570714971903605},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-3.3-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"nova-lite-v1","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":6.899999999999998e-08,"completion":2.7599999999999993e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.020699999999999996,"max_completion_cost":0.0014131199999999997,"max_cost":0.021759839999999996},"sats_pricing":{"prompt":0.00010616783580790298,"completion":0.0004246713432316119,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":31.850350742370896,"max_completion_cost":2.1743172773458532,"max_cost":33.48108870038028},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"amazon/nova-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-micro-v1","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":4.025e-08,"completion":1.61e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005152,"max_completion_cost":0.00082432,"max_cost":0.00577024},"sats_pricing":{"prompt":6.193123755461008e-05,"completion":0.00024772495021844034,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.927198406990092,"max_completion_cost":1.2683517451184145,"max_cost":8.878462215828902},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"amazon/nova-micro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-pro-v1","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":9.199999999999999e-07,"completion":3.6799999999999995e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27599999999999997,"max_completion_cost":0.018841599999999997,"max_cost":0.2901312},"sats_pricing":{"prompt":0.0014155711441053733,"completion":0.005662284576421493,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":424.67134323161196,"max_completion_cost":28.990897031278042,"max_cost":446.41451600507054},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"amazon/nova-pro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4375e-06,"input_cache_write":0.0,"max_prompt_cost":0.368,"max_completion_cost":0.188416,"max_cost":0.509312},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002211829912664646,"input_cache_write":0.0,"max_prompt_cost":566.2284576421494,"max_completion_cost":289.90897031278047,"max_cost":783.6601853767347},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-2024-11-20","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2407","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.30146559999999994,"max_completion_cost":0.9043968,"max_cost":0.9043968},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":463.85435250044867,"max_completion_cost":1391.5630575013463,"max_cost":1391.5630575013463},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-large-2407","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":7.59e-07,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024870912,"max_completion_cost":0.03768319999999999,"max_cost":0.03768319999999999},"sats_pricing":{"prompt":0.001167846193886933,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.26798408128702,"max_completion_cost":57.981794062556084,"max_cost":57.981794062556084},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":4.5999999999999994e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.47103999999999996,"max_completion_cost":0.47103999999999996,"max_cost":0.47103999999999996},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":724.7724257819511,"max_completion_cost":724.7724257819511,"max_cost":724.7724257819511},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":1024000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thedrummer/unslopnemo-12b","alias_ids":null,"forwarded_model_id":null},{"id":"magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.45e-06,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0565248,"max_completion_cost":0.011776,"max_cost":0.061235200000000004},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":86.97269109383414,"max_completion_cost":18.11931064454878,"max_cost":94.22041535165366},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthracite-org/magnum-v4-72b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.1499999999999998e-07,"completion":2.2999999999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0037683199999999995,"max_completion_cost":0.007536639999999999,"max_cost":0.007536639999999999},"sats_pricing":{"prompt":0.00017694639301317166,"completion":0.0003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.798179406255609,"max_completion_cost":11.596358812511218,"max_cost":11.596358812511218},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-2.5-7b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"rocinante-12b","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":2.8749999999999995e-07,"completion":5.749999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018841599999999997,"max_completion_cost":0.03768319999999999,"max_cost":0.03768319999999999},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0008847319650658582,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.990897031278042,"max_completion_cost":57.981794062556084,"max_cost":57.981794062556084},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"thedrummer/rocinante-12b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":3.105e-08,"completion":2.3115e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0018629999999999999,"max_completion_cost":0.013869,"max_cost":0.013869},"sats_pricing":{"prompt":4.777552611355635e-05,"completion":0.0003556622499564751,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.866531566813381,"max_completion_cost":21.339734997388504,"max_cost":21.339734997388504},"per_request_limits":null,"top_provider":{"context_length":60000,"max_completion_tokens":60000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-3.2-1b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.749999999999999e-08,"completion":3.795e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.049741824,"max_cost":0.049741824},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.0005839230969434665,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":76.53596816257404,"max_cost":76.53596816257404},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-3.2-3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":4.14e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.013565952,"max_completion_cost":0.007536639999999999,"max_cost":0.014319615999999999},"sats_pricing":{"prompt":0.000637007014847418,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":20.873445862520192,"max_completion_cost":11.596358812511218,"max_cost":22.033081743771316},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"qwen/qwen-2.5-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-08-2024","name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02208,"max_completion_cost":0.00276,"max_cost":0.024149999999999998},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":33.97370745852896,"max_completion_cost":4.24671343231612,"max_cost":37.15874253276605},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"cohere/command-r-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.368,"max_completion_cost":0.046,"max_cost":0.40249999999999997},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":566.2284576421494,"max_completion_cost":70.77855720526867,"max_cost":619.3123755461008},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"cohere/command-r-plus-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":9.775e-07,"completion":9.775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.12812288,"max_completion_cost":0.01601536,"max_cost":0.12812288},"sats_pricing":{"prompt":0.0015040443406119592,"completion":0.0015040443406119592,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":197.1380998126907,"max_completion_cost":24.64226247658634,"max_cost":197.1380998126907},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"sao10k/l3.1-euryale-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":8.049999999999999e-07,"completion":8.049999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.10551295999999999,"max_completion_cost":0.013189119999999999,"max_cost":0.10551295999999999},"sats_pricing":{"prompt":0.0012386247510922017,"completion":0.0012386247510922017,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":162.34902337515706,"max_completion_cost":20.293627921894632,"max_cost":162.34902337515706},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":1.1499999999999998e-06,"completion":1.1499999999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.018841599999999997,"max_cost":0.15073279999999997},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.0017694639301317164,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":231.92717625022433,"max_completion_cost":28.990897031278042,"max_cost":231.92717625022433},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","alias_ids":null,"forwarded_model_id":null},{"id":"l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.5999999999999995e-08,"completion":5.749999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00037683199999999996,"max_completion_cost":0.00047103999999999994,"max_cost":0.00047103999999999994},"sats_pricing":{"prompt":7.077855720526867e-05,"completion":8.847319650658583e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5798179406255609,"max_completion_cost":0.7247724257819511,"max_cost":0.7247724257819511},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"sao10k/l3-lunaris-8b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4375e-06,"input_cache_write":0.0,"max_prompt_cost":0.368,"max_completion_cost":0.188416,"max_cost":0.509312},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002211829912664646,"input_cache_write":0.0,"max_prompt_cost":566.2284576421494,"max_completion_cost":289.90897031278047,"max_cost":783.6601853767347},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-2024-08-06","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.5999999999999994e-07,"completion":4.5999999999999994e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06029311999999999,"max_completion_cost":0.007536639999999999,"max_cost":0.06029311999999999},"sats_pricing":{"prompt":0.0007077855720526866,"completion":0.0007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":92.77087050008974,"max_completion_cost":11.596358812511218,"max_cost":92.77087050008974},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-3.1-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.749999999999999e-08,"completion":9.199999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8749999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.007536639999999999,"max_completion_cost":0.012058623999999999,"max_cost":0.012058623999999999},"sats_pricing":{"prompt":8.847319650658583e-05,"completion":0.00014155711441053733,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4236598253292915e-05,"input_cache_write":0.0,"max_prompt_cost":11.596358812511218,"max_completion_cost":18.55417410001795,"max_cost":18.55417410001795},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"meta-llama/llama-3.1-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-nemo","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.185e-08,"completion":3.449999999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0028639232,"max_completion_cost":0.0005652479999999999,"max_cost":0.0030711808},"sats_pricing":{"prompt":3.361981467250262e-05,"completion":5.308391790395149e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.406616348754263,"max_completion_cost":0.8697269109383412,"max_cost":4.725516216098322},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-nemo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini","name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.02208,"max_completion_cost":0.01130496,"max_cost":0.030558719999999998},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00013270979475987876,"input_cache_write":0.0,"max_prompt_cost":33.97370745852896,"max_completion_cost":17.39453821876683,"max_cost":47.01961112260408},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.725e-07,"completion":6.9e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.02208,"max_completion_cost":0.01130496,"max_cost":0.030558719999999998},"sats_pricing":{"prompt":0.00026541958951975753,"completion":0.0010616783580790301,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00013270979475987876,"input_cache_write":0.0,"max_prompt_cost":33.97370745852896,"max_completion_cost":17.39453821876683,"max_cost":47.01961112260408},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-mini-2024-07-18","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini:batch","name":"OpenAI: GPT-4o-mini (batch)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.625e-08,"completion":3.45e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":4.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.01104,"max_completion_cost":0.00565248,"max_cost":0.015279359999999999},"sats_pricing":{"prompt":0.00013270979475987876,"completion":0.0005308391790395151,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":6.635489737993938e-05,"input_cache_write":0.0,"max_prompt_cost":16.98685372926448,"max_completion_cost":8.697269109383415,"max_cost":23.50980556130204},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":7.474999999999999e-07,"completion":7.474999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0061235199999999995,"max_completion_cost":0.0015308799999999999,"max_cost":0.0061235199999999995},"sats_pricing":{"prompt":0.0011501515545856158,"completion":0.0011501515545856158,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.422041535165365,"max_completion_cost":2.355510383791341,"max_cost":9.422041535165365},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"google/gemma-2-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o","name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.875e-06,"completion":1.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4375e-06,"input_cache_write":0.0,"max_prompt_cost":0.368,"max_completion_cost":0.188416,"max_cost":0.509312},"sats_pricing":{"prompt":0.004423659825329292,"completion":0.017694639301317167,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002211829912664646,"input_cache_write":0.0,"max_prompt_cost":566.2284576421494,"max_completion_cost":289.90897031278047,"max_cost":783.6601853767347},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.736,"max_completion_cost":0.070656,"max_cost":0.783104},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1132.4569152842987,"max_completion_cost":108.71586386729267,"max_cost":1204.934157862494},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o-2024-05-13","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o:batch","name":"OpenAI: GPT-4o (batch)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4375e-06,"completion":5.75e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":7.1875e-07,"input_cache_write":0.0,"max_prompt_cost":0.184,"max_completion_cost":0.094208,"max_cost":0.254656},"sats_pricing":{"prompt":0.002211829912664646,"completion":0.008847319650658584,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.001105914956332323,"input_cache_write":0.0,"max_prompt_cost":283.1142288210747,"max_completion_cost":144.95448515639023,"max_cost":391.83009268836736},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.15073279999999997,"max_completion_cost":0.4521984,"max_cost":0.4521984},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":231.92717625022433,"max_completion_cost":695.7815287506731,"max_cost":695.7815287506731},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mixtral-8x22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"wizardlm-2-8x22b","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":7.129999999999999e-07,"completion":7.129999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04672645499999999,"max_completion_cost":0.005703999999999999,"max_cost":0.04672645499999999},"sats_pricing":{"prompt":0.0010970676366816642,"completion":0.0010970676366816642,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":71.89632756993286,"max_completion_cost":8.776541093453314,"max_cost":71.89632756993286},"per_request_limits":null,"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"microsoft/wizardlm-2-8x22b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.472,"max_completion_cost":0.141312,"max_cost":1.566208},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2264.9138305685974,"max_completion_cost":217.43172773458534,"max_cost":2409.868315724988},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo:batch","name":"OpenAI: GPT-4 Turbo (batch)","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.75e-06,"completion":1.725e-05,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.736,"max_completion_cost":0.070656,"max_cost":0.783104},"sats_pricing":{"prompt":0.008847319650658584,"completion":0.02654195895197575,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1132.4569152842987,"max_completion_cost":108.71586386729267,"max_cost":1204.934157862494},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"claude-3-haiku","name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":1.4375e-06,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":3.449999999999999e-08,"input_cache_write":3.45e-07,"max_prompt_cost":0.05749999999999999,"max_completion_cost":0.005888,"max_cost":0.062210399999999985},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.002211829912664646,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":5.308391790395149e-05,"input_cache_write":0.0005308391790395151,"max_prompt_cost":88.47319650658582,"max_completion_cost":9.05965532227439,"max_cost":95.72092076440534},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"anthropic/claude-3-haiku","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large","name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.2999999999999996e-06,"completion":6.9e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2999999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.29439999999999994,"max_completion_cost":0.8832,"max_cost":0.8832},"sats_pricing":{"prompt":0.003538927860263433,"completion":0.0106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003538927860263433,"input_cache_write":0.0,"max_prompt_cost":452.9827661137194,"max_completion_cost":1358.9482983411585,"max_cost":1358.9482983411585},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mistralai/mistral-large","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1499999999999998e-06,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004709249999999999,"max_completion_cost":0.009418499999999998,"max_cost":0.009418499999999998},"sats_pricing":{"prompt":0.0017694639301317164,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.2459547938893785,"max_completion_cost":14.491909587778757,"max_cost":14.491909587778757},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-3.5-turbo-0613","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.15e-05,"completion":3.45e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.472,"max_completion_cost":0.141312,"max_cost":1.566208},"sats_pricing":{"prompt":0.017694639301317167,"completion":0.0530839179039515,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2264.9138305685974,"max_completion_cost":217.43172773458534,"max_cost":2409.868315724988},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4-turbo-preview","alias_ids":null,"forwarded_model_id":null},{"id":"auto","name":"Auto Router","created":1699401600,"description":"Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...","context_length":2000000,"architecture":{"modality":"text+image+file+audio+video->text+image","input_modalities":["text","image","audio","file","video"],"output_modalities":["text","image"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":-1.15,"completion":-1.15,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-2300000.0,"max_completion_cost":-2300000.0,"max_cost":-2300000.0},"sats_pricing":{"prompt":-1769.4639301317166,"completion":-1769.4639301317166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-3538927860.2634335,"max_completion_cost":-3538927860.2634335,"max_cost":0.001},"per_request_limits":null,"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openrouter/auto","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":1.725e-06,"completion":2.2999999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007063875,"max_completion_cost":0.009418499999999998,"max_cost":0.009418499999999998},"sats_pricing":{"prompt":0.002654195895197575,"completion":0.003538927860263433,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.86893219083407,"max_completion_cost":14.491909587778757,"max_cost":14.491909587778757},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-3.5-turbo-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.45e-06,"completion":4.599999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05652825,"max_completion_cost":0.018841599999999997,"max_cost":0.06123864999999999},"sats_pricing":{"prompt":0.00530839179039515,"completion":0.007077855720526866,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":86.97799948562454,"max_completion_cost":28.990897031278042,"max_cost":94.22572374344404},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-3.5-turbo-16k","alias_ids":null,"forwarded_model_id":null},{"id":"weaver","name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":5.749999999999999e-07,"completion":8.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004599999999999999,"max_completion_cost":0.005175,"max_cost":0.006325},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.0013270979475987876,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.077855720526865,"max_completion_cost":7.962587685592726,"max_cost":9.732051615724442},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":6000,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"mancer/weaver","alias_ids":null,"forwarded_model_id":null},{"id":"remm-slerp-l2-13b","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":5.174999999999999e-07,"completion":7.474999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0031795199999999995,"max_completion_cost":0.004592639999999999,"max_cost":0.004592639999999999},"sats_pricing":{"prompt":0.0007962587685592724,"completion":0.0011501515545856158,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.89221387402817,"max_completion_cost":7.066531151374023,"max_cost":7.066531151374023},"per_request_limits":null,"top_provider":{"context_length":6144,"max_completion_tokens":6144,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"undi95/remm-slerp-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"mythomax-l2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":9.199999999999999e-08,"completion":1.265e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00037683199999999996,"max_completion_cost":0.000518144,"max_cost":0.000518144},"sats_pricing":{"prompt":0.00014155711441053733,"completion":0.00019464103231448885,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5798179406255609,"max_completion_cost":0.7972496683601463,"max_cost":0.7972496683601463},"per_request_limits":null,"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"base","canonical_slug":"gryphe/mythomax-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo","name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.749999999999999e-07,"completion":1.725e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.009421374999999997,"max_completion_cost":0.0070656,"max_cost":0.014131775},"sats_pricing":{"prompt":0.0008847319650658582,"completion":0.002654195895197575,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14.496333247604086,"max_completion_cost":10.871586386729268,"max_cost":21.7440575054236},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo:batch","name":"OpenAI: GPT-3.5 Turbo (batch)","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8749999999999995e-07,"completion":8.625e-07,"request":0.0,"image":0.0,"web_search":0.0115,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004710687499999999,"max_completion_cost":0.0035328,"max_cost":0.0070658875},"sats_pricing":{"prompt":0.0004423659825329291,"completion":0.0013270979475987876,"request":0.001,"image":0.0,"web_search":17.694639301317167,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.248166623802043,"max_completion_cost":5.435793193364634,"max_cost":10.8720287527118},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4","name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.45e-05,"completion":6.9e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2825895,"max_completion_cost":0.282624,"max_cost":0.4239015},"sats_pricing":{"prompt":0.0530839179039515,"completion":0.106167835807903,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":434.81037155126677,"max_completion_cost":434.8634554691707,"max_cost":652.2420992858521},"per_request_limits":null,"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"base","canonical_slug":"openai/gpt-4","alias_ids":null,"forwarded_model_id":null}]}