{"data":[{"id":"muse-spark-1.2","name":"Meta: Muse Spark 1.2","created":1785959287,"description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":4.4625e-06,"request":0.0,"image":0.0,"web_search":0.002625,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":1.3762560000000001,"max_completion_cost":4.6792704,"max_cost":4.6792704},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.006863997559877708,"request":0.001,"image":0.0,"web_search":4.037645623457475,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":2116.889148631273,"max_completion_cost":7197.423105346327,"max_cost":7197.423105346327},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta/muse-spark-1.2-20260805","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.8-max","name":"Qwen: Qwen3.8 Max","created":1785731612,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":2.6250000000000003e-06,"max_prompt_cost":2.0999999999999996,"max_completion_cost":0.8257536000000001,"max_cost":2.6505023999999997},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.004037645623457476,"max_prompt_cost":3230.1164987659795,"max_completion_cost":1270.1334891787637,"max_cost":4076.872158218489},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.8-max-20260803","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","created":1785606009,"description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":9.407999999999999e-08,"completion":1.8815999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.8816e-08,"input_cache_write":0.0,"max_prompt_cost":0.09865003007999999,"max_completion_cost":0.024662507519999998,"max_cost":0.11098128384},"sats_pricing":{"prompt":0.0001447092191447159,"completion":0.0002894184382894318,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8941843828943185e-05,"input_cache_write":0.0,"max_prompt_cost":151.73861417388963,"max_completion_cost":37.93465354347241,"max_cost":170.70594094562583},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~deepseek/deepseek-v4-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":9.45e-08,"completion":1.89e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.89e-08,"input_cache_write":0.0,"max_prompt_cost":0.099090432,"max_completion_cost":0.072576,"max_cost":0.135378432},"sats_pricing":{"prompt":0.00014535524244446913,"completion":0.00029071048488893826,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.9071048488893822e-05,"input_cache_write":0.0,"max_prompt_cost":152.41601870145166,"max_completion_cost":111.63282619735227,"max_cost":208.23243180012776},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-flash-20260731","alias_ids":null,"forwarded_model_id":null},{"id":"inkling-small","name":"Thinking Machines: Inkling Small","created":1785443117,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.725e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.24772608,"max_completion_cost":0.33030144,"max_cost":0.45416448},"sats_pricing":{"prompt":0.0007267762122223455,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":381.0400467536291,"max_completion_cost":508.0533956715055,"max_cost":698.57341904832},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-small-20260730","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-flash","name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.15e-08,"completion":1.365e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-09,"input_cache_write":3.9900000000000007e-08,"max_prompt_cost":0.0315,"max_completion_cost":0.008945664,"max_cost":0.03838128},"sats_pricing":{"prompt":4.8451747481489696e-05,"completion":0.00020995757241978874,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.69034949629794e-06,"input_cache_write":6.137221347655364e-05,"max_prompt_cost":48.451747481489704,"max_completion_cost":13.759779466103275,"max_cost":59.03619322464606},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-flash-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5-fast","name":"Claude Opus 5 (Fast)","created":1784912546,"description":"Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":5.25e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-06,"input_cache_write":1.3125e-05,"max_prompt_cost":10.500000000000002,"max_completion_cost":6.720000000000001,"max_cost":15.876000000000001},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.08075291246914951,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00161505824938299,"input_cache_write":0.020188228117287377,"max_prompt_cost":16150.582493829905,"max_completion_cost":10336.372796051137,"max_cost":24419.68073067081},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-fast-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5","name":"Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5:batch","name":"Claude Opus 5 (batch)","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":3.28125e-06,"max_prompt_cost":2.6250000000000004,"max_completion_cost":1.6800000000000002,"max_cost":3.9690000000000003},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.005047057029321844,"max_prompt_cost":4037.645623457476,"max_completion_cost":2584.093199012784,"max_cost":6104.920182667703},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"ling-3.0-flash","name":"Ling-3.0-flash","created":1784818580,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.205e-08,"completion":6.615e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.41e-09,"input_cache_write":0.0,"max_prompt_cost":0.0057802752,"max_completion_cost":0.0021676032,"max_cost":0.007225344},"sats_pricing":{"prompt":3.391622323704279e-05,"completion":0.00010174866971112837,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.783244647408559e-06,"input_cache_write":0.0,"max_prompt_cost":8.890934424251345,"max_completion_cost":3.3341004090942543,"max_cost":11.113668030314182},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-3.0-flash-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-s-2.1","name":"Poolside: Laguna S 2.1","created":1784652683,"description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.45e-08,"completion":1.89e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.45e-09,"input_cache_write":0.0,"max_prompt_cost":0.099090432,"max_completion_cost":0.024772608,"max_cost":0.111476736},"sats_pricing":{"prompt":0.00014535524244446913,"completion":0.00029071048488893826,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4535524244446911e-05,"input_cache_write":0.0,"max_prompt_cost":152.41601870145166,"max_completion_cost":38.104004675362916,"max_cost":171.4680210391331},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"poolside/laguna-s-2.1-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash","name":"Google: Gemini 3.6 Flash","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":7.875e-06,"request":0.0,"image":1.5750000000000002e-06,"web_search":0.014700000000000001,"internal_reasoning":7.875e-06,"input_cache_read":1.575e-07,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":1.6515072000000002,"max_completion_cost":0.516096,"max_cost":2.064384},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.012112936870372426,"request":0.001,"image":0.0024225873740744853,"web_search":22.610815491361862,"internal_reasoning":0.012112936870372426,"input_cache_read":0.00024225873740744852,"input_cache_write":0.00013458818744858247,"max_prompt_cost":2540.2669783575275,"max_completion_cost":793.8334307367273,"max_cost":3175.3337229469093},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash:batch","name":"Google: Gemini 3.6 Flash (batch)","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.875000000000001e-07,"completion":3.9375e-06,"request":0.0,"image":7.875000000000001e-07,"web_search":0.014700000000000001,"internal_reasoning":3.9375e-06,"input_cache_read":7.875e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.8257536000000001,"max_completion_cost":0.258048,"max_cost":1.032192},"sats_pricing":{"prompt":0.0012112936870372426,"completion":0.006056468435186213,"request":0.001,"image":0.0012112936870372426,"web_search":22.610815491361862,"internal_reasoning":0.006056468435186213,"input_cache_read":0.00012112936870372426,"input_cache_write":0.00013458818744858247,"max_prompt_cost":1270.1334891787637,"max_completion_cost":396.91671536836367,"max_cost":1587.6668614734547},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite","name":"Google: Gemini 3.5 Flash Lite","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":3.15e-07,"web_search":0.014700000000000001,"internal_reasoning":2.6250000000000003e-06,"input_cache_read":3.15e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.33030144,"max_completion_cost":0.17203200000000002,"max_cost":0.4816896},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.004037645623457476,"request":0.001,"image":0.00048451747481489705,"web_search":22.610815491361862,"internal_reasoning":0.004037645623457476,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":508.0533956715055,"max_completion_cost":264.61114357890915,"max_cost":740.9112020209454},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite:batch","name":"Google: Gemini 3.5 Flash Lite (batch)","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":1.575e-07,"web_search":0.014700000000000001,"internal_reasoning":1.3125000000000001e-06,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.08601600000000001,"max_cost":0.2408448},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.002018822811728738,"request":0.001,"image":0.00024225873740744852,"web_search":22.610815491361862,"internal_reasoning":0.002018822811728738,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":132.30557178945458,"max_cost":370.4556010104727},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"longcat-2.0","name":"Meituan: LongCat 2.0","created":1784554658,"description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-09,"input_cache_write":0.0,"max_prompt_cost":0.33035814,"max_completion_cost":0.33030144,"max_cost":0.57808422},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.69034949629794e-06,"input_cache_write":0.0,"max_prompt_cost":508.1406088169722,"max_completion_cost":508.0533956715055,"max_cost":889.1806555706013},"per_request_limits":null,"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meituan/longcat-2.0-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"inkling","name":"Thinking Machines: Inkling","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.975e-07,"completion":4.2525e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6800000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.52297728,"max_completion_cost":1.11476736,"max_cost":1.3762560000000001},"sats_pricing":{"prompt":0.0015343053369138405,"completion":0.00654098591000111,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025840931990127845,"input_cache_write":0.0,"max_prompt_cost":804.4178764798836,"max_completion_cost":1714.680210391331,"max_cost":2116.889148631273},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"inkling:batch","name":"Thinking Machines: Inkling (batch)","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":2.12625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.925e-08,"input_cache_write":0.0,"max_prompt_cost":0.2752512,"max_completion_cost":1.11476736,"max_cost":1.11476736},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.003270492955000555,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00013727995119755415,"input_cache_write":0.0,"max_prompt_cost":423.3778297262545,"max_completion_cost":1714.680210391331,"max_cost":1714.680210391331},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k3","name":"MoonshotAI: Kimi K3","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":0.0,"max_prompt_cost":3.3030144000000004,"max_completion_cost":16.515072,"max_cost":16.515072},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.0,"max_prompt_cost":5080.533956715055,"max_completion_cost":25402.669783575275,"max_cost":25402.669783575275},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k3-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.1","name":"Meta: Muse Spark 1.1","created":1784215741,"description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":4.4625e-06,"request":0.0,"image":0.0,"web_search":0.002625,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":1.3762560000000001,"max_completion_cost":4.6792704,"max_cost":4.6792704},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.006863997559877708,"request":0.001,"image":0.0,"web_search":4.037645623457475,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":2116.889148631273,"max_completion_cost":7197.423105346327,"max_cost":7197.423105346327},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta/muse-spark-1.1-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-air-v2.5","name":"Kwaipilot: KAT-Coder-Air V2.5","created":1783714590,"description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.04032,"max_completion_cost":0.0504,"max_cost":0.07812},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":62.01823677630682,"max_completion_cost":77.52279597038353,"max_cost":120.16033375409445},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-air-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","created":1783714589,"description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.77e-07,"completion":3.108e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":0.198912,"max_completion_cost":0.24864,"max_cost":0.385392},"sats_pricing":{"prompt":0.0011951431045434126,"completion":0.0047805724181736505,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":305.95663476311364,"max_completion_cost":382.44579345389207,"max_cost":592.7909798535327},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-pro-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":1.3125e-07,"max_prompt_cost":0.11025,"max_completion_cost":0.08064,"max_cost":0.17745},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.00020188228117287374,"max_prompt_cost":169.58111618521397,"max_completion_cost":124.03647355261364,"max_cost":272.9448441457253},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro:batch","name":"OpenAI: GPT-5.6 Luna Pro (batch)","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.11025,"max_completion_cost":0.08064,"max_cost":0.17745},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":169.58111618521397,"max_completion_cost":124.03647355261364,"max_cost":272.9448441457253},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna","name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":1.3125e-07,"max_prompt_cost":0.11025,"max_completion_cost":0.08064,"max_cost":0.17745},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.00020188228117287374,"max_prompt_cost":169.58111618521397,"max_completion_cost":124.03647355261364,"max_cost":272.9448441457253},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna:batch","name":"OpenAI: GPT-5.6 Luna (batch)","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.11025,"max_completion_cost":0.08064,"max_cost":0.17745},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":169.58111618521397,"max_completion_cost":124.03647355261364,"max_cost":272.9448441457253},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":1.3125000000000001e-06,"max_prompt_cost":1.1024999999999998,"max_completion_cost":0.8064000000000001,"max_cost":1.7745000000000002},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.002018822811728738,"max_prompt_cost":1695.8111618521393,"max_completion_cost":1240.3647355261367,"max_cost":2729.4484414572535},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro:batch","name":"OpenAI: GPT-5.6 Terra Pro (batch)","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":1.1024999999999998,"max_completion_cost":0.8064000000000001,"max_cost":1.7745000000000002},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":1695.8111618521393,"max_completion_cost":1240.3647355261367,"max_cost":2729.4484414572535},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra","name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":1.3125000000000001e-06,"max_prompt_cost":1.1024999999999998,"max_completion_cost":0.8064000000000001,"max_cost":1.7745000000000002},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.002018822811728738,"max_prompt_cost":1695.8111618521393,"max_completion_cost":1240.3647355261367,"max_cost":2729.4484414572535},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra:batch","name":"OpenAI: GPT-5.6 Terra (batch)","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":1.1024999999999998,"max_completion_cost":0.8064000000000001,"max_cost":1.7745000000000002},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":1695.8111618521393,"max_completion_cost":1240.3647355261367,"max_cost":2729.4484414572535},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.5125,"max_completion_cost":4.032,"max_cost":8.8725},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8479.055809260699,"max_completion_cost":6201.823677630682,"max_cost":13647.242207286266},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro:batch","name":"OpenAI: GPT-5.6 Sol Pro (batch)","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":2.75625,"max_completion_cost":2.016,"max_cost":4.43625},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":4239.527904630349,"max_completion_cost":3100.911838815341,"max_cost":6823.621103643133},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol","name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.5125,"max_completion_cost":4.032,"max_cost":8.8725},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8479.055809260699,"max_completion_cost":6201.823677630682,"max_cost":13647.242207286266},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol:batch","name":"OpenAI: GPT-5.6 Sol (batch)","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":2.75625,"max_completion_cost":2.016,"max_cost":4.43625},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":4239.527904630349,"max_completion_cost":3100.911838815341,"max_cost":6823.621103643133},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.5","name":"SpaceXAI: Grok 4.5","created":1783523154,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":0.0,"max_prompt_cost":1.0499999999999998,"max_completion_cost":3.1500000000000004,"max_cost":3.1500000000000004},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.0,"max_prompt_cost":1615.0582493829897,"max_completion_cost":4845.174748148971,"max_cost":4845.174748148971},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.5-20260708","alias_ids":null,"forwarded_model_id":null},{"id":"grok-latest","name":"xAI: Grok Latest","created":1783519360,"description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":0.0,"max_prompt_cost":1.0499999999999998,"max_completion_cost":3.1500000000000004,"max_cost":3.1500000000000004},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.0,"max_prompt_cost":1615.0582493829897,"max_completion_cost":4845.174748148971,"max_cost":4845.174748148971},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~x-ai/grok-latest","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","created":1783443096,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.35e-07,"completion":1.47e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.89e-07,"input_cache_write":0.0,"max_prompt_cost":0.09633792,"max_completion_cost":0.04816896,"max_cost":0.1204224},"sats_pricing":{"prompt":0.001130540774568093,"completion":0.002261081549136186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00029071048488893826,"input_cache_write":0.0,"max_prompt_cost":148.1822404041891,"max_completion_cost":74.09112020209454,"max_cost":185.22780050523636},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-3.0-mini-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0","name":"AionLabs: Aion-3.0","created":1783443095,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.875000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.41287680000000004,"max_completion_cost":0.20643840000000002,"max_cost":0.5160960000000001},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0012112936870372426,"input_cache_write":0.0,"max_prompt_cost":635.0667445893819,"max_completion_cost":317.53337229469093,"max_cost":793.8334307367275},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-3.0-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"hy3","name":"Tencent: Hy3","created":1783344048,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3859999999999998e-07,"completion":5.543999999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.4649999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.036333158399999996,"max_completion_cost":0.07096319999999999,"max_cost":0.08955555839999998},"sats_pricing":{"prompt":0.00021318768891855468,"completion":0.0008527507556742187,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.329692222963867e-05,"input_cache_write":0.0,"max_prompt_cost":55.8858735238656,"max_completion_cost":109.15209672629999,"max_cost":137.7499460685906},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hy3-20260706","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","created":1783002429,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-08,"completion":1.26e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.016515072,"max_completion_cost":0.004128768,"max_cost":0.018579456},"sats_pricing":{"prompt":9.690349496297939e-05,"completion":0.00019380698992595879,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":25.40266978357527,"max_completion_cost":6.3506674458938175,"max_cost":28.578003506522183},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"poolside/laguna-xs-2.1-20260625","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":2.6250000000000003e-06,"max_prompt_cost":2.0999999999999996,"max_completion_cost":1.344,"max_cost":3.1752},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.004037645623457476,"max_prompt_cost":3230.1164987659795,"max_completion_cost":2067.2745592102274,"max_cost":4883.936146134161},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5:batch","name":"Anthropic: Claude Sonnet 5 (batch)","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":1.3125000000000001e-06,"max_prompt_cost":1.0499999999999998,"max_completion_cost":0.672,"max_cost":1.5876},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.002018822811728738,"max_prompt_cost":1615.0582493829897,"max_completion_cost":1033.6372796051137,"max_cost":2441.9680730670807},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","created":1782837225,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.014700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0172032,"max_completion_cost":0.10321920000000001,"max_cost":0.10321920000000001},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0024225873740744853,"request":0.001,"image":0.0,"web_search":22.610815491361862,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":26.461114357890906,"max_completion_cost":158.76668614734547,"max_cost":158.76668614734547},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","created":1782312964,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.625e-08,"completion":1.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.6250000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.02752512,"max_cost":0.02752512},"sats_pricing":{"prompt":4.037645623457475e-05,"completion":0.000161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.037645623457476e-06,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":42.337782972625455,"max_cost":42.337782972625455},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nex-agi/nex-n2-mini","alias_ids":null,"forwarded_model_id":null},{"id":"fugu-ultra","name":"Sakana: Fugu Ultra","created":1782276303,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":0.0,"max_prompt_cost":5.250000000000001,"max_completion_cost":4.032,"max_cost":8.61},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.0,"max_prompt_cost":8075.291246914952,"max_completion_cost":6201.823677630682,"max_cost":13243.477644940518},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sakana/fugu-ultra-20260615","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","created":1781754065,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":3.1500000000000003e-06,"request":0.0,"image":0.0,"web_search":0.014700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0688128,"max_completion_cost":0.10321920000000001,"max_cost":0.1548288},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0048451747481489706,"request":0.001,"image":0.0,"web_search":22.610815491361862,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":105.84445743156363,"max_completion_cost":158.76668614734547,"max_cost":238.15002922101817},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","created":1781754054,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.2600000000000001e-05,"request":0.0,"image":2.1e-06,"web_search":0.014700000000000001,"internal_reasoning":1.2600000000000001e-05,"input_cache_read":2.1e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.1376256,"max_completion_cost":0.41287680000000004,"max_cost":0.48168960000000005},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.019380698992595882,"request":0.001,"image":0.00323011649876598,"web_search":22.610815491361862,"internal_reasoning":0.019380698992595882,"input_cache_read":0.000323011649876598,"input_cache_write":0.0006056468435186213,"max_prompt_cost":211.68891486312725,"max_completion_cost":635.0667445893819,"max_cost":740.9112020209456},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-pro-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1609000000000002e-07,"completion":6.7914e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.0131000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.22127616000000003,"max_completion_cost":0.08692992000000001,"max_cost":0.28054656000000006},"sats_pricing":{"prompt":0.0003323789877230194,"completion":0.001044619675700918,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.172752629141789e-05,"input_cache_write":0.0,"max_prompt_cost":340.35608342837185,"max_completion_cost":133.7113184897175,"max_cost":431.5228914895429},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2:batch","name":"Z.ai: GLM 5.2 (batch)","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":512000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.35e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.365e-07,"input_cache_write":0.0,"max_prompt_cost":0.37632,"max_completion_cost":1.1827200000000002,"max_cost":1.1827200000000002},"sats_pricing":{"prompt":0.001130540774568093,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00020995757241978874,"input_cache_write":0.0,"max_prompt_cost":578.8368765788637,"max_completion_cost":1819.2016121050003,"max_cost":1819.2016121050003},"per_request_limits":null,"top_provider":{"context_length":512000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code","name":"MoonshotAI: Kimi K2.7 Code","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.35e-07,"completion":3.675e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":0.19267584,"max_completion_cost":0.9633792,"max_cost":0.9633792},"sats_pricing":{"prompt":0.001130540774568093,"completion":0.005652703872840465,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":296.3644808083782,"max_completion_cost":1481.8224040418909,"max_cost":1481.8224040418909},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code:batch","name":"MoonshotAI: Kimi K2.7 Code (batch)","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.9875e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.975e-08,"input_cache_write":0.0,"max_prompt_cost":0.13074432,"max_completion_cost":0.5505024,"max_cost":0.5505024},"sats_pricing":{"prompt":0.0007671526684569202,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015343053369138405,"input_cache_write":0.0,"max_prompt_cost":201.1044691199709,"max_completion_cost":846.755659452509,"max_cost":846.755659452509},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-latest","name":"Anthropic: Claude Fable Latest","created":1781029944,"description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":5.25e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-06,"input_cache_write":1.3125e-05,"max_prompt_cost":10.500000000000002,"max_completion_cost":6.720000000000001,"max_cost":15.876000000000001},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.08075291246914951,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00161505824938299,"input_cache_write":0.020188228117287377,"max_prompt_cost":16150.582493829905,"max_completion_cost":10336.372796051137,"max_cost":24419.68073067081},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-fable-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5","name":"Anthropic: Claude Fable 5","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":5.25e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-06,"input_cache_write":1.3125e-05,"max_prompt_cost":10.500000000000002,"max_completion_cost":6.720000000000001,"max_cost":15.876000000000001},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.08075291246914951,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00161505824938299,"input_cache_write":0.020188228117287377,"max_prompt_cost":16150.582493829905,"max_completion_cost":10336.372796051137,"max_cost":24419.68073067081},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5:batch","name":"Anthropic: Claude Fable 5 (batch)","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","created":1780937140,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.0688128,"max_completion_cost":0.2752512,"max_cost":0.2752512},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":105.84445743156363,"max_completion_cost":423.3778297262545,"max_cost":423.3778297262545},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nex-agi/nex-n2-pro","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":3.78e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.32274144,"max_completion_cost":1.9364486399999998,"max_cost":1.9364486399999998},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.005814209697778764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":496.4249762759479,"max_completion_cost":2978.5498576556874,"max_cost":2978.5498576556874},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b:batch","name":"NVIDIA: Nemotron 3 Ultra (batch)","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.89e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.16137072,"max_completion_cost":0.9682243199999999,"max_cost":0.9682243199999999},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.002907104848889382,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":248.21248813797396,"max_completion_cost":1489.2749288278437,"max_cost":1489.2749288278437},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.3600000000000004e-07,"completion":1.3440000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.72e-08,"input_cache_write":4.2e-07,"max_prompt_cost":0.336,"max_completion_cost":0.17616076800000002,"max_cost":0.46812057600000007},"sats_pricing":{"prompt":0.0005168186398025569,"completion":0.0020672745592102276,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010336372796051138,"input_cache_write":0.000646023299753196,"max_prompt_cost":516.8186398025568,"max_completion_cost":270.96181102480296,"max_cost":720.0399980711592},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-plus-20260602","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.64512,"max_cost":0.64899072},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.690349496297939e-05,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":992.2917884209091,"max_cost":998.2455391514346},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3:batch","name":"MiniMax: MiniMax M3 (batch)","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":524288,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.08257536,"max_completion_cost":0.33030144,"max_cost":0.33030144},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":127.01334891787637,"max_completion_cost":508.0533956715055,"max_cost":508.0533956715055},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.7-flash","name":"StepFun: Step 3.7 Flash","created":1779985069,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":1.2075e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2000000000000006e-08,"input_cache_write":0.0,"max_prompt_cost":0.05376,"max_completion_cost":0.30912,"max_cost":0.30912},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0018573169867904388,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.460232997531961e-05,"input_cache_write":0.0,"max_prompt_cost":82.6909823684091,"max_completion_cost":475.4731486183523,"max_cost":475.4731486183523},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":256000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"stepfun/step-3.7-flash-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","created":1779913703,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":5.25e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-06,"input_cache_write":1.3125e-05,"max_prompt_cost":10.500000000000002,"max_completion_cost":6.720000000000001,"max_cost":15.876000000000001},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.08075291246914951,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00161505824938299,"input_cache_write":0.020188228117287377,"max_prompt_cost":16150.582493829905,"max_completion_cost":10336.372796051137,"max_cost":24419.68073067081},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-fast-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8","name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8:batch","name":"Anthropic: Claude Opus 4.8 (batch)","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":3.28125e-06,"max_prompt_cost":2.6250000000000004,"max_completion_cost":1.6800000000000002,"max_cost":3.9690000000000003},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.005047057029321844,"max_prompt_cost":4037.645623457476,"max_completion_cost":2584.093199012784,"max_cost":6104.920182667703},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-max","name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.5487500000000002e-06,"completion":4.64625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.0975e-07,"input_cache_write":1.9359375e-06,"max_prompt_cost":1.5487500000000003,"max_completion_cost":0.60899328,"max_cost":1.9547455200000001},"sats_pricing":{"prompt":0.002382210917839911,"completion":0.007146632753519731,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00047644218356798204,"input_cache_write":0.002977763647299888,"max_prompt_cost":2382.210917839911,"max_completion_cost":936.7234482693382,"max_cost":3006.693216686136},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.7-max-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"grok-build-0.1","name":"SpaceXAI: Grok Build 0.1","created":1779298123,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":0.5376,"max_cost":0.5376},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":413.4549118420454,"max_completion_cost":826.9098236840908,"max_cost":826.9098236840908},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-build-0.1-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash","name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":9.450000000000001e-06,"request":0.0,"image":1.5750000000000002e-06,"web_search":0.014700000000000001,"internal_reasoning":9.450000000000001e-06,"input_cache_read":1.575e-07,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":1.6515072000000002,"max_completion_cost":0.6193152000000001,"max_cost":2.1676032000000003},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.014535524244446912,"request":0.001,"image":0.0024225873740744853,"web_search":22.610815491361862,"internal_reasoning":0.014535524244446912,"input_cache_read":0.00024225873740744852,"input_cache_write":0.00013458818744858247,"max_prompt_cost":2540.2669783575275,"max_completion_cost":952.6001168840728,"max_cost":3334.100409094255},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash:batch","name":"Google: Gemini 3.5 Flash (batch)","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.875000000000001e-07,"completion":4.7250000000000005e-06,"request":0.0,"image":7.875000000000001e-07,"web_search":0.014700000000000001,"internal_reasoning":4.7250000000000005e-06,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.8257536000000001,"max_completion_cost":0.30965760000000003,"max_cost":1.0838016000000001},"sats_pricing":{"prompt":0.0012112936870372426,"completion":0.007267762122223456,"request":0.001,"image":0.0012112936870372426,"web_search":22.610815491361862,"internal_reasoning":0.007267762122223456,"input_cache_read":0.00012112936870372426,"input_cache_write":0.0,"max_prompt_cost":1270.1334891787637,"max_completion_cost":476.3000584420364,"max_cost":1667.0502045471276},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.15e-05,"completion":0.00015749999999999998,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.1500000000000003e-06,"input_cache_write":3.9374999999999995e-05,"max_prompt_cost":31.5,"max_completion_cost":20.159999999999997,"max_cost":47.628},"sats_pricing":{"prompt":0.048451747481489706,"completion":0.2422587374074485,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0048451747481489706,"input_cache_write":0.06056468435186212,"max_prompt_cost":48451.74748148971,"max_completion_cost":31009.118388153405,"max_cost":73259.04219201244},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"perceptron-mk1","name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00516096,"max_completion_cost":0.012902400000000001,"max_cost":0.016773120000000002},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0024225873740744853,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.938334307367273,"max_completion_cost":19.845835768418183,"max_cost":25.79958649894364},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perceptron/perceptron-mk1-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.875e-08,"completion":6.562500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.02064384,"max_completion_cost":0.043008000000000005,"max_cost":0.05849088000000001},"sats_pricing":{"prompt":0.00012112936870372426,"completion":0.001009411405864369,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":31.753337229469093,"max_completion_cost":66.15278589472729,"max_cost":89.96778881682911},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ring-2.6-1t-20260508","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":2.625e-07,"web_search":0.014700000000000001,"internal_reasoning":1.5750000000000002e-06,"input_cache_read":2.625e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.2752512,"max_completion_cost":0.10321920000000001,"max_cost":0.3612672},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0024225873740744853,"request":0.001,"image":0.0004037645623457475,"web_search":22.610815491361862,"internal_reasoning":0.0024225873740744853,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":423.3778297262545,"max_completion_cost":158.76668614734547,"max_cost":555.6834015157091},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite:batch","name":"Google: Gemini 3.1 Flash Lite (batch)","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.3125e-07,"completion":7.875000000000001e-07,"request":0.0,"image":1.3125e-07,"web_search":0.014700000000000001,"internal_reasoning":7.875000000000001e-07,"input_cache_read":1.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1376256,"max_completion_cost":0.051609600000000005,"max_cost":0.1806336},"sats_pricing":{"prompt":0.00020188228117287374,"completion":0.0012112936870372426,"request":0.001,"image":0.00020188228117287374,"web_search":22.610815491361862,"internal_reasoning":0.0012112936870372426,"input_cache_read":2.0188228117287376e-05,"input_cache_write":0.0,"max_prompt_cost":211.68891486312725,"max_completion_cost":79.38334307367273,"max_cost":277.84170075785454},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-chat-latest","name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":0.0,"max_prompt_cost":2.1,"max_completion_cost":4.032,"max_cost":5.46},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.0,"max_prompt_cost":3230.1164987659804,"max_completion_cost":6201.823677630682,"max_cost":8398.302896791549},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-chat-latest-20260505","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.3","name":"SpaceXAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":1.3125000000000002,"max_completion_cost":2.6250000000000004,"max_cost":2.6250000000000004},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":2018.822811728738,"max_completion_cost":4037.645623457476,"max_cost":4037.645623457476},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.3-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":1.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.01376256,"max_cost":0.01376256},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":21.168891486312727,"max_cost":21.168891486312727},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":7.875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.41287680000000004,"max_completion_cost":2.064384,"max_cost":2.064384},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.012112936870372426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":635.0667445893819,"max_completion_cost":3175.3337229469093,"max_cost":3175.3337229469093},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3.5-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-latest","name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":1.3125000000000001e-06,"max_prompt_cost":0.21,"max_completion_cost":0.336,"max_cost":0.4788},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.002018822811728738,"max_prompt_cost":323.011649876598,"max_completion_cost":516.8186398025568,"max_cost":736.4665617186434},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-haiku-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-mini-latest","name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":7.875000000000001e-07,"completion":4.7250000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.31500000000000006,"max_completion_cost":0.6048000000000001,"max_cost":0.8190000000000002},"sats_pricing":{"prompt":0.0012112936870372426,"completion":0.007267762122223456,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00012112936870372426,"input_cache_write":0.0,"max_prompt_cost":484.5174748148971,"max_completion_cost":930.2735516446024,"max_cost":1259.7454345187325},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~openai/gpt-mini-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-pro-latest","name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.2600000000000001e-05,"request":0.0,"image":2.1e-06,"web_search":0.014700000000000001,"internal_reasoning":1.2600000000000001e-05,"input_cache_read":2.1e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":2.2020096,"max_completion_cost":0.8257536000000001,"max_cost":2.8901376},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.019380698992595882,"request":0.001,"image":0.00323011649876598,"web_search":22.610815491361862,"internal_reasoning":0.019380698992595882,"input_cache_read":0.000323011649876598,"input_cache_write":0.0006056468435186213,"max_prompt_cost":3387.022637810036,"max_completion_cost":1270.1334891787637,"max_cost":4445.467212125673},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~google/gemini-pro-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-latest","name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.045e-07,"input_cache_write":0.0,"max_prompt_cost":2.7525120000000003,"max_completion_cost":15.4140672,"max_cost":15.4140672},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00046836689232106714,"input_cache_write":0.0,"max_prompt_cost":4233.778297262546,"max_completion_cost":23709.158464670254,"max_cost":23709.158464670254},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":1048576,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~moonshotai/kimi-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-flash-latest","name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":7.875e-06,"request":0.0,"image":1.5750000000000002e-06,"web_search":0.014700000000000001,"internal_reasoning":7.875e-06,"input_cache_read":1.575e-07,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":1.6515072000000002,"max_completion_cost":0.516096,"max_cost":2.064384},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.012112936870372426,"request":0.001,"image":0.0024225873740744853,"web_search":22.610815491361862,"internal_reasoning":0.012112936870372426,"input_cache_read":0.00024225873740744852,"input_cache_write":0.00013458818744858247,"max_prompt_cost":2540.2669783575275,"max_completion_cost":793.8334307367273,"max_cost":3175.3337229469093},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~google/gemini-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":2.6250000000000003e-06,"max_prompt_cost":2.0999999999999996,"max_completion_cost":1.344,"max_cost":3.1752},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.004037645623457476,"max_prompt_cost":3230.1164987659795,"max_completion_cost":2067.2745592102274,"max_cost":4883.936146134161},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-sonnet-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-latest","name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.5125,"max_completion_cost":4.032,"max_cost":8.8725},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8479.055809260699,"max_completion_cost":6201.823677630682,"max_cost":13647.242207286266},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~openai/gpt-latest","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.89e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.315,"max_completion_cost":0.12386304,"max_cost":0.41821919999999996},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.002907104848889382,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0006056468435186213,"max_prompt_cost":484.517474814897,"max_completion_cost":190.52002337681455,"max_cost":643.2841609622425},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-plus-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-flash","name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.9687500000000002e-07,"completion":1.1812500000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.4609375e-07,"max_prompt_cost":0.19687500000000002,"max_completion_cost":0.07741440000000001,"max_cost":0.26138700000000004},"sats_pricing":{"prompt":0.00030282342175931066,"completion":0.001816940530555864,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0003785292771991383,"max_prompt_cost":302.8234217593107,"max_completion_cost":119.0750146105091,"max_cost":402.0526006014016},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.2752512,"max_cost":0.2752512},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":423.3778297262545,"max_cost":423.3778297262545},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-max-preview","name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.0783500000000002e-06,"completion":6.4701e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":1.3479375000000001e-06,"max_prompt_cost":0.28268298240000006,"max_completion_cost":0.4240244736,"max_cost":0.6360367104},"sats_pricing":{"prompt":0.0016586648221163312,"completion":0.009951988932697985,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0020733310276454136,"max_prompt_cost":434.8090311288635,"max_completion_cost":652.2135466932951,"max_cost":978.3203200399427},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-max-preview-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-27b","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":3.78e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.26e-07,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.99090432,"max_cost":0.99090432},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.005814209697778764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00019380698992595879,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":1524.1601870145164,"max_cost":1524.1601870145164},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-27b-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.15e-05,"completion":0.000189,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.1500000000000003e-06,"input_cache_write":0.0,"max_prompt_cost":33.075,"max_completion_cost":24.192,"max_cost":53.235},"sats_pricing":{"prompt":0.048451747481489706,"completion":0.29071048488893825,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0048451747481489706,"input_cache_write":0.0,"max_prompt_cost":50874.33485556419,"max_completion_cost":37210.94206578409,"max_cost":81883.4532437176},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro:batch","name":"OpenAI: GPT-5.5 Pro (batch)","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":9.45e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.5375,"max_completion_cost":12.096,"max_cost":26.6175},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.14535524244446912,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25437.167427782097,"max_completion_cost":18605.471032892045,"max_cost":40941.7266218588},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":0.0,"max_prompt_cost":5.5125,"max_completion_cost":4.032,"max_cost":8.8725},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.0,"max_prompt_cost":8479.055809260699,"max_completion_cost":6201.823677630682,"max_cost":13647.242207286266},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5:batch","name":"OpenAI: GPT-5.5 (batch)","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":2.75625,"max_completion_cost":2.016,"max_cost":4.43625},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":4239.527904630349,"max_completion_cost":3100.911838815341,"max_cost":6823.621103643133},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":4.5675000000000006e-07,"completion":9.135000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.8062500000000006e-09,"input_cache_write":0.0,"max_prompt_cost":0.47893708800000007,"max_completion_cost":0.35078400000000004,"max_cost":0.6543290880000001},"sats_pricing":{"prompt":0.0007025503384816008,"completion":0.0014051006769632017,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.85458615401334e-06,"input_cache_write":0.0,"max_prompt_cost":736.6774237236831,"max_completion_cost":539.5586599538694,"max_cost":1006.4567537006178},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.47e-07,"completion":2.94e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.9400000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.154140672,"max_completion_cost":0.115605504,"max_cost":0.211943424},"sats_pricing":{"prompt":0.00022610815491361862,"completion":0.00045221630982723724,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5221630982723725e-05,"input_cache_write":0.0,"max_prompt_cost":237.09158464670256,"max_completion_cost":177.8186884850269,"max_cost":326.00092888921597},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v4-flash-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.875e-08,"completion":6.562500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.02064384,"max_completion_cost":0.021504000000000002,"max_cost":0.03956736},"sats_pricing":{"prompt":0.00012112936870372426,"completion":0.001009411405864369,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":31.753337229469093,"max_completion_cost":33.076392947363644,"max_cost":60.8605630231491},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-2.6-1t-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.615e-08,"completion":2.2050000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.205e-08,"input_cache_write":0.0,"max_prompt_cost":0.0173408256,"max_completion_cost":0.057802752000000006,"max_cost":0.057802752000000006},"sats_pricing":{"prompt":0.00010174866971112837,"completion":0.00033916223237042797,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.391622323704279e-05,"input_cache_write":0.0,"max_prompt_cost":26.672803272754035,"max_completion_cost":88.90934424251347,"max_cost":88.90934424251347},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hy3-preview-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.5675000000000006e-07,"completion":9.135000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.78e-09,"input_cache_write":0.0,"max_prompt_cost":0.47893708800000007,"max_completion_cost":0.11973427200000002,"max_cost":0.538804224},"sats_pricing":{"prompt":0.0007025503384816008,"completion":0.0014051006769632017,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.814209697778764e-06,"input_cache_write":0.0,"max_prompt_cost":736.6774237236831,"max_completion_cost":184.16935593092077,"max_cost":828.7621016891434},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.47e-07,"completion":2.94e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.9399999999999998e-09,"input_cache_write":0.0,"max_prompt_cost":0.154140672,"max_completion_cost":0.038535168,"max_cost":0.173408256},"sats_pricing":{"prompt":0.00022610815491361862,"completion":0.00045221630982723724,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.522163098272372e-06,"input_cache_write":0.0,"max_prompt_cost":237.09158464670256,"max_completion_cost":59.27289616167564,"max_cost":266.7280327275404},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"xiaomi/mimo-v2.5-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.4e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.1e-06,"input_cache_write":0.0,"max_prompt_cost":2.2847999999999997,"max_completion_cost":2.016,"max_cost":3.2256},"sats_pricing":{"prompt":0.01292046599506392,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00323011649876598,"input_cache_write":0.0,"max_prompt_cost":3514.366750657386,"max_completion_cost":3100.911838815341,"max_cost":4961.458942104546},"per_request_limits":null,"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-image-2-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":3.15e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-09,"input_cache_write":0.0,"max_prompt_cost":0.0027525120000000004,"max_completion_cost":0.001032192,"max_cost":0.0034406400000000005},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":4.8451747481489696e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.2301164987659805e-06,"input_cache_write":0.0,"max_prompt_cost":4.233778297262546,"max_completion_cost":1.5876668614734544,"max_cost":5.292222871578183},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inclusionai/ling-2.6-flash-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-latest","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"~anthropic/claude-opus-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.08475e-07,"completion":2.562e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0248e-07,"input_cache_write":0.0,"max_prompt_cost":0.1595080704,"max_completion_cost":0.671612928,"max_cost":0.671612928},"sats_pricing":{"prompt":0.0009359262555174428,"completion":0.003940742128494496,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015762968513977985,"input_cache_write":0.0,"max_prompt_cost":245.34745232636453,"max_completion_cost":1033.0419045320612,"max_cost":1033.0419045320612},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.6-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7","name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7:batch","name":"Anthropic: Claude Opus 4.7 (batch)","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":3.28125e-06,"max_prompt_cost":2.6250000000000004,"max_completion_cost":1.6800000000000002,"max_cost":3.9690000000000003},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.005047057029321844,"max_prompt_cost":4037.645623457476,"max_completion_cost":2584.093199012784,"max_cost":6104.920182667703},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.996e-07,"completion":3.1416e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.8564000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.20267089919999998,"max_completion_cost":0.4117757952,"max_cost":0.48342712320000003},"sats_pricing":{"prompt":0.0015375354534126065,"completion":0.004832254282153906,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00028554229849091266,"input_cache_write":0.0,"max_prompt_cost":311.73838825031277,"max_completion_cost":633.3732332704768,"max_cost":743.5837745710925},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.1-20260406","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":7.35e-08,"completion":3.57e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019267584,"max_completion_cost":0.005849088,"max_cost":0.023912448},"sats_pricing":{"prompt":0.00011305407745680931,"completion":0.0005491198047902166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":29.63644808083782,"max_completion_cost":8.996778881682909,"max_cost":36.78094895746836},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-31b-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":3.57e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.093585408,"max_cost":0.093585408},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0005491198047902166,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":143.94846210692654,"max_cost":143.94846210692654},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-4-31b-it-20260402","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-plus","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.4125e-07,"completion":2.0475e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.265625e-07,"max_prompt_cost":0.34125,"max_completion_cost":0.13418496,"max_cost":0.4530708},"sats_pricing":{"prompt":0.0005248939310494718,"completion":0.0031493635862968306,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0006561174138118397,"max_prompt_cost":524.8939310494718,"max_completion_cost":206.3966919915491,"max_cost":696.8911743757627},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.6-plus-04-02","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.26e-06,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.52e-07,"input_cache_write":0.0,"max_prompt_cost":0.25546752,"max_completion_cost":0.5505024,"max_cost":0.6408191999999999},"sats_pricing":{"prompt":0.0019380698992595882,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00038761397985191757,"input_cache_write":0.0,"max_prompt_cost":392.94754821468,"max_completion_cost":846.755659452509,"max_cost":985.6765098314363},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5v-turbo-20260401","alias_ids":null,"forwarded_model_id":null},{"id":"trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":8.925e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.060555264000000004,"max_completion_cost":0.23396352,"max_cost":0.23396352},"sats_pricing":{"prompt":0.0003553128148642579,"completion":0.0013727995119755417,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.690349496297939e-05,"input_cache_write":0.0,"max_prompt_cost":93.14312253977602,"max_completion_cost":359.8711552673164,"max_cost":359.8711552673164},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"arcee-ai/trinity-large-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":2.6250000000000004,"max_completion_cost":5.250000000000001,"max_cost":5.250000000000001},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":4037.645623457476,"max_completion_cost":8075.291246914952,"max_cost":8075.291246914952},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20","name":"SpaceXAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":2.6250000000000004,"max_completion_cost":5.250000000000001,"max_cost":5.250000000000001},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":4037.645623457476,"max_completion_cost":8075.291246914952,"max_cost":8075.291246914952},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"x-ai/grok-4.20-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.08064,"max_completion_cost":0.1008,"max_cost":0.15624},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.690349496297939e-05,"input_cache_write":0.0,"max_prompt_cost":124.03647355261364,"max_completion_cost":155.04559194076705,"max_cost":240.3206675081889},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","alias_ids":null,"forwarded_model_id":null},{"id":"reka-edge","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":1.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00172032,"max_completion_cost":0.00172032,"max_cost":0.00172032},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.646111435789091,"max_completion_cost":2.646111435789091,"max_cost":2.646111435789091},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"rekaai/reka-edge-2603","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.064512,"max_completion_cost":0.16515072,"max_cost":0.18837504},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.690349496297939e-05,"input_cache_write":0.0,"max_prompt_cost":99.22917884209092,"max_completion_cost":254.02669783575274,"max_cost":289.74920221890545},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.7-20260318","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.1000000000000003e-08,"input_cache_write":0.0,"max_prompt_cost":0.084,"max_completion_cost":0.168,"max_cost":0.22512000000000001},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.002018822811728738,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":3.230116498765981e-05,"input_cache_write":0.0,"max_prompt_cost":129.2046599506392,"max_completion_cost":258.4093199012784,"max_cost":346.2684886677131},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano:batch","name":"OpenAI: GPT-5.4 Nano (batch)","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":6.562500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.042,"max_completion_cost":0.084,"max_cost":0.11256000000000001},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.001009411405864369,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":64.6023299753196,"max_completion_cost":129.2046599506392,"max_cost":173.13424433385654},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875000000000001e-07,"completion":4.7250000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.31500000000000006,"max_completion_cost":0.6048000000000001,"max_cost":0.8190000000000002},"sats_pricing":{"prompt":0.0012112936870372426,"completion":0.007267762122223456,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00012112936870372426,"input_cache_write":0.0,"max_prompt_cost":484.5174748148971,"max_completion_cost":930.2735516446024,"max_cost":1259.7454345187325},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini:batch","name":"OpenAI: GPT-5.4 Mini (batch)","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.9375000000000004e-07,"completion":2.3625000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.9375e-08,"input_cache_write":0.0,"max_prompt_cost":0.15750000000000003,"max_completion_cost":0.30240000000000006,"max_cost":0.4095000000000001},"sats_pricing":{"prompt":0.0006056468435186213,"completion":0.003633881061111728,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":6.056468435186213e-05,"input_cache_write":0.0,"max_prompt_cost":242.25873740744856,"max_completion_cost":465.1367758223012,"max_cost":629.8727172593663},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.16515072,"max_cost":0.16515072},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":254.02669783575274,"max_cost":254.02669783575274},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-2603","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-turbo","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.26e-06,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.52e-07,"input_cache_write":0.0,"max_prompt_cost":0.25546752,"max_completion_cost":0.5505024,"max_cost":0.6408191999999999},"sats_pricing":{"prompt":0.0019380698992595882,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00038761397985191757,"input_cache_write":0.0,"max_prompt_cost":392.94754821468,"max_completion_cost":846.755659452509,"max_cost":985.6765098314363},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5-turbo-20260315","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.925e-08,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.023396352,"max_completion_cost":0.00688128,"max_cost":0.028815359999999998},"sats_pricing":{"prompt":0.00013727995119755415,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":35.987115526731635,"max_completion_cost":10.584445743156364,"max_cost":44.32236654946727},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0688128,"max_completion_cost":0.2752512,"max_cost":0.3096576},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":105.84445743156363,"max_completion_cost":423.3778297262545,"max_cost":476.30005844203635},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-9b","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":1.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.04128768,"max_cost":0.04128768},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.00024225873740744852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":63.506674458938186,"max_cost":63.506674458938186},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-9b-20260310","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.15e-05,"completion":0.000189,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.1500000000000003e-06,"input_cache_write":0.0,"max_prompt_cost":33.075,"max_completion_cost":24.192,"max_cost":53.235},"sats_pricing":{"prompt":0.048451747481489706,"completion":0.29071048488893825,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0048451747481489706,"input_cache_write":0.0,"max_prompt_cost":50874.33485556419,"max_completion_cost":37210.94206578409,"max_cost":81883.4532437176},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro:batch","name":"OpenAI: GPT-5.4 Pro (batch)","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":9.45e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.5375,"max_completion_cost":12.096,"max_cost":26.6175},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.14535524244446912,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25437.167427782097,"max_completion_cost":18605.471032892045,"max_cost":40941.7266218588},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":2.75625,"max_completion_cost":2.016,"max_cost":4.43625},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":4239.527904630349,"max_completion_cost":3100.911838815341,"max_cost":6823.621103643133},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4:batch","name":"OpenAI: GPT-5.4 (batch)","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":7.875e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-07,"input_cache_write":0.0,"max_prompt_cost":1.378125,"max_completion_cost":1.008,"max_cost":2.218125},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.012112936870372426,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0,"max_prompt_cost":2119.7639523151747,"max_completion_cost":1550.4559194076705,"max_cost":3411.8105518215666},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"mercury-2","name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":7.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.0336,"max_completion_cost":0.03937500000000001,"max_cost":0.05985},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0012112936870372426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":51.68186398025568,"max_completion_cost":60.56468435186214,"max_cost":92.05832021483043},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"inception/mercury-2-20260304","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-chat","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8375e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.8375e-07,"input_cache_write":0.0,"max_prompt_cost":0.2352,"max_completion_cost":0.2408448,"max_cost":0.4459392},"sats_pricing":{"prompt":0.0028263519364202325,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00028263519364202325,"input_cache_write":0.0,"max_prompt_cost":361.7730478617898,"max_completion_cost":370.4556010104727,"max_cost":685.9216987459534},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.3-chat-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":2.625e-07,"web_search":0.014700000000000001,"internal_reasoning":1.5750000000000002e-06,"input_cache_read":2.625e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.2752512,"max_completion_cost":0.10321920000000001,"max_cost":0.3612672},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0024225873740744853,"request":0.001,"image":0.0004037645623457475,"web_search":22.610815491361862,"internal_reasoning":0.0024225873740744853,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":423.3778297262545,"max_completion_cost":158.76668614734547,"max_cost":555.6834015157091},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.05505024,"max_cost":0.06881280000000001},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":84.67556594525091,"max_cost":105.84445743156365},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":3.1500000000000003e-06,"request":0.0,"image":0.0,"web_search":0.014700000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0344064,"max_completion_cost":0.20643840000000002,"max_cost":0.20643840000000002},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0048451747481489706,"request":0.001,"image":0.0,"web_search":22.610815491361862,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":52.92222871578181,"max_completion_cost":317.53337229469093,"max_cost":317.53337229469093},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.47e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.038535168,"max_completion_cost":0.2752512,"max_cost":0.2752512},"sats_pricing":{"prompt":0.00022610815491361862,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":59.27289616167564,"max_completion_cost":423.3778297262545,"max_cost":423.3778297262545},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-27b","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.0475000000000003e-07,"completion":1.6380000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05367398400000001,"max_completion_cost":0.10734796800000002,"max_cost":0.14760345600000002},"sats_pricing":{"prompt":0.0003149363586296831,"completion":0.002519490869037465,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":82.55867679661965,"max_completion_cost":165.1173535932393,"max_cost":227.03636119070404},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-27b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.045e-07,"completion":2.52e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.079822848,"max_completion_cost":0.2064384,"max_cost":0.261316608},"sats_pricing":{"prompt":0.00046836689232106714,"completion":0.0038761397985191764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":122.77957062061382,"max_completion_cost":317.53337229469093,"max_cost":401.94432709636294},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.825e-08,"completion":2.73e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06825,"max_completion_cost":0.017891328,"max_cost":0.08166849600000001},"sats_pricing":{"prompt":0.00010497878620989437,"completion":0.0004199151448395775,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":104.97878620989437,"max_completion_cost":27.51955893220655,"max_cost":125.61845540904928},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-flash-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.2600000000000001e-05,"request":0.0,"image":2.1e-06,"web_search":0.014700000000000001,"internal_reasoning":1.2600000000000001e-05,"input_cache_read":2.1e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":2.2020096,"max_completion_cost":0.8257536000000001,"max_cost":2.8901376},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.019380698992595882,"request":0.001,"image":0.00323011649876598,"web_search":22.610815491361862,"internal_reasoning":0.019380698992595882,"input_cache_read":0.000323011649876598,"input_cache_write":0.0006056468435186213,"max_prompt_cost":3387.022637810036,"max_completion_cost":1270.1334891787637,"max_cost":4445.467212125673},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-codex","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8375e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.8375e-07,"input_cache_write":0.0,"max_prompt_cost":0.735,"max_completion_cost":1.8816,"max_cost":2.3814},"sats_pricing":{"prompt":0.0028263519364202325,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00028263519364202325,"input_cache_write":0.0,"max_prompt_cost":1130.540774568093,"max_completion_cost":2894.1843828943183,"max_cost":3662.952109600622},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.3-codex-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"aion-2.0","name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.4e-07,"completion":1.68e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.11010048,"max_completion_cost":0.05505024,"max_cost":0.13762560000000001},"sats_pricing":{"prompt":0.001292046599506392,"completion":0.002584093199012784,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":169.35113189050182,"max_completion_cost":84.67556594525091,"max_cost":211.6889148631273},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-2.0-20260223","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.2600000000000001e-05,"request":0.0,"image":2.1e-06,"web_search":0.014700000000000001,"internal_reasoning":1.2600000000000001e-05,"input_cache_read":2.1e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":2.2020096,"max_completion_cost":0.8257536000000001,"max_cost":2.8901376},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.019380698992595882,"request":0.001,"image":0.00323011649876598,"web_search":22.610815491361862,"internal_reasoning":0.019380698992595882,"input_cache_read":0.000323011649876598,"input_cache_write":0.0006056468435186213,"max_prompt_cost":3387.022637810036,"max_completion_cost":1270.1334891787637,"max_cost":4445.467212125673},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview:batch","name":"Google: Gemini 3.1 Pro Preview (batch)","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":6.300000000000001e-06,"request":0.0,"image":1.05e-06,"web_search":0.014700000000000001,"internal_reasoning":6.300000000000001e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.1010048,"max_completion_cost":0.41287680000000004,"max_cost":1.4450688},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.009690349496297941,"request":0.001,"image":0.00161505824938299,"web_search":22.610815491361862,"internal_reasoning":0.009690349496297941,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1693.511318905018,"max_completion_cost":635.0667445893819,"max_cost":2222.7336060628363},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":3.9375e-06,"max_prompt_cost":3.1500000000000004,"max_completion_cost":2.016,"max_cost":4.7628},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.006056468435186213,"max_prompt_cost":4845.174748148971,"max_completion_cost":3100.911838815341,"max_cost":7325.904219201244},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6:batch","name":"Anthropic: Claude Sonnet 4.6 (batch)","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":7.875e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":1.96875e-06,"max_prompt_cost":1.5750000000000002,"max_completion_cost":1.008,"max_cost":2.3814},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.012112936870372426,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0030282342175931066,"max_prompt_cost":2422.5873740744855,"max_completion_cost":1550.4559194076705,"max_cost":3662.952109600622},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.73e-07,"completion":1.6380000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.273,"max_completion_cost":0.10734796800000002,"max_cost":0.36245664000000005},"sats_pricing":{"prompt":0.0004199151448395775,"completion":0.002519490869037465,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":419.91514483957747,"max_completion_cost":165.1173535932393,"max_cost":557.5129395006103},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-plus-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.0950000000000006e-07,"completion":2.457e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.10734796800000002,"max_completion_cost":0.161021952,"max_cost":0.241532928},"sats_pricing":{"prompt":0.0006298727172593662,"completion":0.0037792363035561967,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":165.1173535932393,"max_completion_cost":247.6760303898589,"max_cost":371.5140455847884},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":9.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.045416448000000005,"max_completion_cost":0.18579456,"max_cost":0.18579456},"sats_pricing":{"prompt":0.0003553128148642579,"completion":0.001453552424444691,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.0006056468435186213,"max_prompt_cost":69.85734190483201,"max_completion_cost":285.7800350652218,"max_cost":285.7800350652218},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.975e-07,"completion":2.6775e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.204288,"max_completion_cost":0.35094528,"max_cost":0.42448896},"sats_pricing":{"prompt":0.0015343053369138405,"completion":0.004118398535926625,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":314.2257329999546,"max_completion_cost":539.8067329009746,"max_cost":652.9279967809582},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":8.190000000000001e-07,"completion":4.095e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.21469593600000003,"max_completion_cost":0.26836992,"max_cost":0.429391872},"sats_pricing":{"prompt":0.0012597454345187325,"completion":0.006298727172593661,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":330.2347071864786,"max_completion_cost":412.7933839830982,"max_cost":660.4694143729571},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-max-thinking-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":5.250000000000001,"max_completion_cost":3.3600000000000003,"max_cost":7.938000000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":8075.291246914952,"max_completion_cost":5168.186398025568,"max_cost":12209.840365335405},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6:batch","name":"Anthropic: Claude Opus 4.6 (batch)","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":3.28125e-06,"max_prompt_cost":2.6250000000000004,"max_completion_cost":1.6800000000000002,"max_cost":3.9690000000000003},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.005047057029321844,"max_prompt_cost":4037.645623457476,"max_completion_cost":2584.093199012784,"max_cost":6104.920182667703},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.26e-07,"completion":8.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.35e-08,"input_cache_write":0.0,"max_prompt_cost":0.033030144,"max_completion_cost":0.22020096,"max_cost":0.22020096},"sats_pricing":{"prompt":0.00019380698992595879,"completion":0.001292046599506392,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011305407745680931,"input_cache_write":0.0,"max_prompt_cost":50.80533956715054,"max_completion_cost":338.70226378100364,"max_cost":338.70226378100364},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.5-flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.02064384,"max_cost":0.04128768},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":31.753337229469093,"max_cost":63.506674458938186},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"stepfun/step-3.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.985000000000001e-07,"completion":2.9924999999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.975e-08,"input_cache_write":0.0,"max_prompt_cost":0.15689318400000002,"max_completion_cost":0.7844659199999999,"max_cost":0.7844659199999999},"sats_pricing":{"prompt":0.0009205832021483045,"completion":0.004602916010741522,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015343053369138405,"input_cache_write":0.0,"max_prompt_cost":241.32536294396513,"max_completion_cost":1206.6268147198255,"max_cost":1206.6268147198255},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.5-0127","alias_ids":null,"forwarded_model_id":null},{"id":"solar-pro-3","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.02064384,"max_completion_cost":0.08257536,"max_cost":0.08257536},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":31.753337229469093,"max_completion_cost":127.01334891787637,"max_cost":127.01334891787637},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"upstage/solar-pro-3","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2-her","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.02064384,"max_completion_cost":0.00258048,"max_cost":0.0225792},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":31.753337229469093,"max_completion_cost":3.9691671536836366,"max_cost":34.73021259473182},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2-her-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"palmyra-x5","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6552,"max_completion_cost":0.051609600000000005,"max_cost":0.70164864},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1007.7963476149858,"max_completion_cost":79.38334307367273,"max_cost":1079.2413563812913},"per_request_limits":null,"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"writer/palmyra-x5-20250428","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.336,"max_completion_cost":0.17203200000000002,"max_cost":0.46502400000000005},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":516.8186398025568,"max_completion_cost":264.61114357890915,"max_cost":715.2769974867388},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-audio","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio-mini","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":2.52e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08064,"max_completion_cost":0.04128768,"max_cost":0.11160576},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.0038761397985191764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":124.03647355261364,"max_completion_cost":63.506674458938186,"max_cost":171.66647939681727},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-audio-mini","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7-flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-08,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.012773376,"max_completion_cost":0.00688128,"max_cost":0.018622464},"sats_pricing":{"prompt":9.690349496297939e-05,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":19.647377410734,"max_completion_cost":10.584445743156364,"max_cost":28.64415629241691},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.7-flash-20260119","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-codex","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8375e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.8375e-07,"input_cache_write":0.0,"max_prompt_cost":0.735,"max_completion_cost":1.8816,"max_cost":2.3814},"sats_pricing":{"prompt":0.0028263519364202325,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00028263519364202325,"input_cache_write":0.0,"max_prompt_cost":1130.540774568093,"max_completion_cost":2894.1843828943183,"max_cost":3662.952109600622},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-codex-20260114","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.875e-08,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02064384,"max_completion_cost":0.01032192,"max_cost":0.02838528},"sats_pricing":{"prompt":0.00012112936870372426,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":31.753337229469093,"max_completion_cost":15.876668614734546,"max_cost":43.66083869052},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0688128,"max_completion_cost":0.0688128,"max_cost":0.12902399999999997},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":105.84445743156363,"max_completion_cost":105.84445743156363,"max_cost":198.45835768418178},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance-seed/seed-1.6-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.064512,"max_completion_cost":0.16515072,"max_cost":0.18837504},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0006056468435186213,"max_prompt_cost":99.22917884209092,"max_completion_cost":254.02669783575274,"max_cost":289.74920221890545},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":1.8375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.400000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.08515584,"max_completion_cost":0.2408448,"max_cost":0.2709504},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.0028263519364202325,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012920465995063923,"input_cache_write":0.0,"max_prompt_cost":130.98251607155998,"max_completion_cost":370.4556010104727,"max_cost":416.7625511367818},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.7-20251222","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":3.1500000000000003e-06,"request":0.0,"image":5.25e-07,"web_search":0.014700000000000001,"internal_reasoning":3.1500000000000003e-06,"input_cache_read":5.25e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.5505024,"max_completion_cost":0.20643840000000002,"max_cost":0.7225344},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0048451747481489706,"request":0.001,"image":0.000807529124691495,"web_search":22.610815491361862,"internal_reasoning":0.0048451747481489706,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":846.755659452509,"max_completion_cost":317.53337229469093,"max_cost":1111.3668030314182},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview:batch","name":"Google: Gemini 3 Flash Preview (batch)","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":2.625e-07,"web_search":0.014700000000000001,"internal_reasoning":1.5750000000000002e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2752512,"max_completion_cost":0.10321920000000001,"max_cost":0.3612672},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0024225873740744853,"request":0.001,"image":0.0004037645623457475,"web_search":22.610815491361862,"internal_reasoning":0.0024225873740744853,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":423.3778297262545,"max_completion_cost":158.76668614734547,"max_cost":555.6834015157091},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.01376256,"max_completion_cost":0.05505024,"max_cost":0.05505024},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":21.168891486312727,"max_completion_cost":84.67556594525091,"max_cost":84.67556594525091},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8375e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.8375e-07,"input_cache_write":0.0,"max_prompt_cost":0.2352,"max_completion_cost":0.2408448,"max_cost":0.4459392},"sats_pricing":{"prompt":0.0028263519364202325,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00028263519364202325,"input_cache_write":0.0,"max_prompt_cost":361.7730478617898,"max_completion_cost":370.4556010104727,"max_cost":685.9216987459534},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-chat-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.205e-05,"completion":0.0001764,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.82,"max_completion_cost":22.5792,"max_cost":28.5768},"sats_pricing":{"prompt":0.03391622323704279,"completion":0.2713297858963423,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13566.489294817118,"max_completion_cost":34730.21259473182,"max_cost":43955.42531520746},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro:batch","name":"OpenAI: GPT-5.2 Pro (batch)","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1025e-05,"completion":8.82e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.41,"max_completion_cost":11.2896,"max_cost":14.2884},"sats_pricing":{"prompt":0.016958111618521395,"completion":0.13566489294817116,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6783.244647408559,"max_completion_cost":17365.10629736591,"max_cost":21977.71265760373},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8375e-06,"completion":1.47e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.8375e-07,"input_cache_write":0.0,"max_prompt_cost":0.735,"max_completion_cost":1.8816,"max_cost":2.3814},"sats_pricing":{"prompt":0.0028263519364202325,"completion":0.02261081549136186,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00028263519364202325,"input_cache_write":0.0,"max_prompt_cost":1130.540774568093,"max_completion_cost":2894.1843828943183,"max_cost":3662.952109600622},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2:batch","name":"OpenAI: GPT-5.2 (batch)","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.1875e-07,"completion":7.35e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":9.1875e-08,"input_cache_write":0.0,"max_prompt_cost":0.3675,"max_completion_cost":0.9408,"max_cost":1.1907},"sats_pricing":{"prompt":0.0014131759682101163,"completion":0.01130540774568093,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00014131759682101163,"input_cache_write":0.0,"max_prompt_cost":565.2703872840465,"max_completion_cost":1447.0921914471592,"max_cost":1831.476054800311},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"relace-search","name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":3.1500000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":0.40320000000000006,"max_cost":0.5376000000000001},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.0048451747481489706,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":413.4549118420454,"max_completion_cost":620.1823677630683,"max_cost":826.9098236840911},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"relace/relace-search-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6v","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":9.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.7750000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.03096576,"max_cost":0.061931520000000004},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.001453552424444691,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.882820371606447e-05,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":47.63000584420364,"max_cost":95.26001168840727},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.6-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-07,"input_cache_write":0.0,"max_prompt_cost":0.525,"max_completion_cost":1.344,"max_cost":1.701},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0,"max_prompt_cost":807.5291246914951,"max_completion_cost":2067.2745592102274,"max_cost":2616.394364000444},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-codex-max-20251204","alias_ids":null,"forwarded_model_id":null},{"id":"nova-2-lite-v1","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.315,"max_completion_cost":0.172029375,"max_cost":0.46638585},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":484.517474814897,"max_completion_cost":264.60710593328565,"max_cost":717.3717280361884},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-2-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1000000000000003e-08,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.05505024,"max_cost":0.05505024},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.230116498765981e-05,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":84.67556594525091,"max_cost":84.67556594525091},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-14b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.04128768,"max_cost":0.04128768},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.00024225873740744852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.4225873740744848e-05,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":63.506674458938186,"max_cost":63.506674458938186},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-8b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":1.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.01376256,"max_completion_cost":0.01376256,"max_cost":0.01376256},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":21.168891486312727,"max_completion_cost":21.168891486312727,"max_cost":21.168891486312727},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/ministral-3b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2512","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.1376256,"max_completion_cost":0.41287680000000004,"max_cost":0.41287680000000004},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0024225873740744853,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.0,"max_prompt_cost":211.68891486312725,"max_completion_cost":635.0667445893819,"max_cost":635.0667445893819},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large-2512","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.8245e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.41225e-07,"input_cache_write":0.0,"max_prompt_cost":0.046276608,"max_completion_cost":0.02752512,"max_cost":0.0552910848},"sats_pricing":{"prompt":0.0004344506690840243,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021722533454201216,"input_cache_write":0.0,"max_prompt_cost":71.18039762272655,"max_completion_cost":42.337782972625455,"max_cost":85.04602154626139},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.2-20251201","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":2.625e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":6.5625e-06,"max_prompt_cost":1.05,"max_completion_cost":1.6800000000000002,"max_cost":2.394},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.040376456234574754,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.010094114058643688,"max_prompt_cost":1615.0582493829902,"max_completion_cost":2584.093199012784,"max_cost":3682.3328085932176},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5:batch","name":"Anthropic: Claude Opus 4.5 (batch)","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":3.28125e-06,"max_prompt_cost":0.525,"max_completion_cost":0.8400000000000001,"max_cost":1.197},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.005047057029321844,"max_prompt_cost":807.5291246914951,"max_completion_cost":1292.046599506392,"max_cost":1841.1664042966088},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":5.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01032192,"max_completion_cost":0.0344064,"max_cost":0.0344064},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.000807529124691495,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":15.876668614734546,"max_completion_cost":52.92222871578181,"max_cost":52.92222871578181},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"allenai/olmo-3-32b-think-20251121","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":1.2600000000000001e-05,"request":0.0,"image":2.1e-06,"web_search":0.014700000000000001,"internal_reasoning":1.2600000000000001e-05,"input_cache_read":2.1e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.1376256,"max_completion_cost":0.41287680000000004,"max_cost":0.48168960000000005},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.019380698992595882,"request":0.001,"image":0.00323011649876598,"web_search":22.610815491361862,"internal_reasoning":0.019380698992595882,"input_cache_read":0.000323011649876598,"input_cache_write":0.0006056468435186213,"max_prompt_cost":211.68891486312725,"max_completion_cost":635.0667445893819,"max_cost":740.9112020209456},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-pro-image-preview-20251120","alias_ids":null,"forwarded_model_id":null},{"id":"cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.3125000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.168,"max_completion_cost":0.168,"max_cost":0.168},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.002018822811728738,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":258.4093199012784,"max_completion_cost":258.4093199012784,"max_cost":258.4093199012784},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-07,"input_cache_write":0.0,"max_prompt_cost":0.525,"max_completion_cost":1.344,"max_cost":1.701},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0,"max_prompt_cost":807.5291246914951,"max_completion_cost":2067.2745592102274,"max_cost":2616.394364000444},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1:batch","name":"OpenAI: GPT-5.1 (batch)","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.562500000000001e-07,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":6.5625e-08,"input_cache_write":0.0,"max_prompt_cost":0.2625,"max_completion_cost":0.672,"max_cost":0.8505},"sats_pricing":{"prompt":0.001009411405864369,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00010094114058643687,"input_cache_write":0.0,"max_prompt_cost":403.76456234574755,"max_completion_cost":1033.6372796051137,"max_cost":1308.197182000222},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.365e-07,"input_cache_write":0.0,"max_prompt_cost":0.525,"max_completion_cost":1.344,"max_cost":1.701},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00020995757241978874,"input_cache_write":0.0,"max_prompt_cost":807.5291246914951,"max_completion_cost":2067.2745592102274,"max_cost":2616.394364000444},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-codex-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.105,"max_completion_cost":0.2688,"max_cost":0.34019999999999995},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":161.505824938299,"max_completion_cost":413.4549118420454,"max_cost":523.2788728000887},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.26342400000000005,"max_cost":0.36535296000000006},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":405.18581360520466,"max_cost":561.9679161757083},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","alias_ids":null,"forwarded_model_id":null},{"id":"nova-premier-v1","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.3125e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.562500000000001e-07,"input_cache_write":0.0,"max_prompt_cost":2.6250000000000004,"max_completion_cost":0.42000000000000004,"max_cost":2.9610000000000003},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.020188228117287377,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.001009411405864369,"input_cache_write":0.0,"max_prompt_cost":4037.645623457476,"max_completion_cost":646.023299753196,"max_cost":4554.464263260033},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-premier-v1","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro-search","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0189,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6300000000000001,"max_completion_cost":0.126,"max_cost":0.7308000000000001},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":29.071048488893823,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":969.0349496297943,"max_completion_cost":193.80698992595882,"max_cost":1124.0805415705613},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-pro-search","alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.00336,"max_completion_cost":0.01008,"max_cost":0.01008},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":5.1681863980255685,"max_completion_cost":15.504559194076705,"max_cost":15.504559194076705},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/voxtral-small-24b-2507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-08,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.9375e-08,"input_cache_write":0.0,"max_prompt_cost":0.01032192,"max_completion_cost":0.02064384,"max_cost":0.0258048},"sats_pricing":{"prompt":0.00012112936870372426,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.056468435186213e-05,"input_cache_write":0.0,"max_prompt_cost":15.876668614734546,"max_completion_cost":31.753337229469093,"max_cost":39.69167153683637},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-oss-safeguard-20b","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.6775e-07,"completion":1.071e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":0.0548352,"max_completion_cost":0.140378112,"max_cost":0.16011878400000001},"sats_pricing":{"prompt":0.0004118398535926625,"completion":0.00164735941437065,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0006056468435186213,"max_prompt_cost":84.34480201577728,"max_completion_cost":215.92269316038983,"max_cost":246.28682188606967},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m2","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.0920000000000001e-07,"completion":4.3680000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.014313062400000002,"max_completion_cost":0.014313062400000002,"max_cost":0.0250478592},"sats_pricing":{"prompt":0.000167966057935831,"completion":0.000671864231743324,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.01564714576524,"max_completion_cost":22.01564714576524,"max_cost":38.527382505089165},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.785e-08,"completion":1.1760000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0023383500000000003,"max_completion_cost":0.0154056,"max_cost":0.0154056},"sats_pricing":{"prompt":2.7455990239510833e-05,"completion":0.0001808865239308949,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.5967347213759195,"max_completion_cost":23.69613463494723,"max_cost":23.69613463494723},"per_request_limits":null,"top_provider":{"context_length":131000,"max_completion_tokens":131000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ibm-granite/granite-4.0-h-micro","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":1.05,"max_completion_cost":0.2688,"max_cost":0.9828000000000001},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":1615.0582493829902,"max_completion_cost":413.4549118420454,"max_cost":1511.6945214224788},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-image-mini","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":1.3125000000000001e-06,"max_prompt_cost":0.21,"max_completion_cost":0.336,"max_cost":0.4788},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.002018822811728738,"max_prompt_cost":323.011649876598,"max_completion_cost":516.8186398025568,"max_cost":736.4665617186434},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5:batch","name":"Anthropic: Claude Haiku 4.5 (batch)","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":6.562500000000001e-07,"max_prompt_cost":0.105,"max_completion_cost":0.168,"max_cost":0.2394},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.001009411405864369,"max_prompt_cost":161.505824938299,"max_completion_cost":258.4093199012784,"max_cost":368.2332808593217},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.89e-07,"completion":2.205e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024772608,"max_completion_cost":0.07225344,"max_cost":0.090832896},"sats_pricing":{"prompt":0.00029071048488893826,"completion":0.0033916223237042795,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.104004675362916,"max_completion_cost":111.13668030314183,"max_cost":139.714683809664},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-8b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.2285000000000002e-07,"completion":4.7775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.016102195200000002,"max_completion_cost":0.015654912,"max_cost":0.027731558400000002},"sats_pricing":{"prompt":0.00018896181517780988,"completion":0.0007348515034692605,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24.767603038985897,"max_completion_cost":24.07961406568073,"max_cost":42.65531634492015},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125000000000001e-06,"input_cache_write":0.0,"max_prompt_cost":4.2,"max_completion_cost":1.344,"max_cost":4.2},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.002018822811728738,"input_cache_write":0.0,"max_prompt_cost":6460.232997531961,"max_completion_cost":2067.2745592102274,"max_cost":6460.232997531961},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-image","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":3.15e-07,"web_search":0.014700000000000001,"internal_reasoning":2.6250000000000003e-06,"input_cache_read":3.15e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.01032192,"max_completion_cost":0.021504000000000002,"max_cost":0.029245440000000004},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.004037645623457476,"request":0.001,"image":0.00048451747481489705,"web_search":22.610815491361862,"internal_reasoning":0.004037645623457476,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":15.876668614734546,"max_completion_cost":33.076392947363644,"max_cost":44.98389440841456},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-image","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":2.52e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.08257536,"max_cost":0.1032192},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0038761397985191764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":127.01334891787637,"max_cost":158.76668614734547},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.01032192,"max_cost":0.049029119999999995},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":15.876668614734546,"max_cost":75.41417591998909},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":0.000126,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.3,"max_completion_cost":16.128,"max_cost":20.412},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.19380698992595882,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9690.34949629794,"max_completion_cost":24807.29471052273,"max_cost":31396.732368005327},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro:batch","name":"OpenAI: GPT-5 Pro (batch)","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-06,"completion":6.3e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.15,"max_completion_cost":8.064,"max_cost":10.206},"sats_pricing":{"prompt":0.012112936870372426,"completion":0.09690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4845.17474814897,"max_completion_cost":12403.647355261364,"max_cost":15698.366184002663},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.10644479999999999,"max_completion_cost":0.2752512,"max_cost":0.3128832},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":163.72814508944998,"max_completion_cost":423.3778297262545,"max_cost":481.26151738414086},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.6","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":3.9375e-06,"max_prompt_cost":3.1500000000000004,"max_completion_cost":1.008,"max_cost":3.9564000000000004},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.006056468435186213,"max_prompt_cost":4845.174748148971,"max_completion_cost":1550.4559194076705,"max_cost":6085.5394836751075},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5:batch","name":"Anthropic: Claude Sonnet 4.5 (batch)","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.5750000000000002e-06,"completion":7.875e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":1.96875e-06,"max_prompt_cost":1.5750000000000002,"max_completion_cost":0.504,"max_cost":1.9782000000000002},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.012112936870372426,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0030282342175931066,"max_prompt_cost":2422.5873740744855,"max_completion_cost":775.2279597038353,"max_cost":3042.7697418375537},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.835e-07,"completion":4.305e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04644864,"max_completion_cost":0.028213248,"max_cost":0.056082432},"sats_pricing":{"prompt":0.00043606572733340734,"completion":0.0006621738822470259,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":71.44500876630545,"max_completion_cost":43.39622754694109,"max_cost":86.26323280672437},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.2-exp","alias_ids":null,"forwarded_model_id":null},{"id":"cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":5.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.575e-07,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.0688128,"max_cost":0.0688128},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.000807529124691495,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024225873740744852,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":105.84445743156363,"max_cost":105.84445743156363},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/cydonia-24b-v4.1","alias_ids":null,"forwarded_model_id":null},{"id":"relace-apply-3","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.925e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.22848000000000002,"max_completion_cost":0.168,"max_cost":0.28224000000000005},"sats_pricing":{"prompt":0.0013727995119755417,"completion":0.002018822811728738,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":351.43667506573865,"max_completion_cost":258.4093199012784,"max_cost":434.1276574341478},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"relace/relace-apply-3","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.1376256,"max_cost":0.17891327999999998},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":211.68891486312725,"max_cost":275.1955893220654},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2050000000000002e-07,"completion":1.995e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.028901376000000003,"max_completion_cost":0.06537216,"max_cost":0.087048192},"sats_pricing":{"prompt":0.00033916223237042797,"completion":0.003068610673827681,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":44.454672121256735,"max_completion_cost":100.55223455998545,"max_cost":133.893238650928},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.190000000000001e-07,"completion":4.095e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.638e-07,"input_cache_write":1.02375e-06,"max_prompt_cost":0.21469593600000003,"max_completion_cost":0.26836992,"max_cost":0.429391872},"sats_pricing":{"prompt":0.0012597454345187325,"completion":0.006298727172593661,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025194908690374647,"input_cache_write":0.0015746817931484153,"max_prompt_cost":330.2347071864786,"max_completion_cost":412.7933839830982,"max_cost":660.4694143729571},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-max","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.825e-07,"completion":3.4125e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.365e-07,"input_cache_write":8.53125e-07,"max_prompt_cost":0.6825,"max_completion_cost":0.2236416,"max_cost":0.8614132800000001},"sats_pricing":{"prompt":0.0010497878620989436,"completion":0.005248939310494718,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00020995757241978874,"input_cache_write":0.0013122348276236795,"max_prompt_cost":1049.7878620989436,"max_completion_cost":343.9944866525818,"max_cost":1324.983451421009},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-plus","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-codex:batch","name":"OpenAI: GPT-5 Codex (batch)","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.562500000000001e-07,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":6.5625e-08,"input_cache_write":0.0,"max_prompt_cost":0.2625,"max_completion_cost":0.672,"max_cost":0.8505},"sats_pricing":{"prompt":0.001009411405864369,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00010094114058643687,"input_cache_write":0.0,"max_prompt_cost":403.76456234574755,"max_completion_cost":1033.6372796051137,"max_cost":1308.197182000222},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-codex","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.835e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4175e-07,"input_cache_write":0.0,"max_prompt_cost":0.037158912,"max_completion_cost":0.0344064,"max_cost":0.062275583999999995},"sats_pricing":{"prompt":0.00043606572733340734,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021803286366670367,"input_cache_write":0.0,"max_prompt_cost":57.156007013044366,"max_completion_cost":52.92222871578181,"max_cost":95.78923397556508},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-v3.1-terminus","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.0475000000000003e-07,"completion":1.02375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.095e-08,"input_cache_write":2.559375e-07,"max_prompt_cost":0.20475000000000004,"max_completion_cost":0.06709248,"max_cost":0.25842398400000005},"sats_pricing":{"prompt":0.0003149363586296831,"completion":0.0015746817931484153,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.298727172593662e-05,"input_cache_write":0.0003936704482871038,"max_prompt_cost":314.9363586296831,"max_completion_cost":103.19834599577455,"max_cost":397.49503542630276},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.33030144,"max_cost":0.33030144},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":508.0533956715055,"max_cost":508.0533956715055},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.45e-08,"completion":1.1550000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024772608,"max_completion_cost":0.018923520000000003,"max_cost":0.042147840000000006},"sats_pricing":{"prompt":0.00014535524244446913,"completion":0.0017765640743212894,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.104004675362916,"max_completion_cost":29.107225793680005,"max_cost":64.82973017683274},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.73e-07,"completion":8.190000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.273,"max_completion_cost":0.026836992000000004,"max_cost":0.290891328},"sats_pricing":{"prompt":0.0004199151448395775,"completion":0.0012597454345187325,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":419.91514483957747,"max_completion_cost":41.279338398309825,"max_cost":447.434703771784},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":5.25e-07,"max_prompt_cost":0.42,"max_completion_cost":0.04128768,"max_cost":0.44752512},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.000807529124691495,"max_prompt_cost":646.023299753196,"max_completion_cost":63.506674458938186,"max_cost":688.3610827258215},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.26342400000000005,"max_cost":0.36535296000000006},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":405.18581360520466,"max_cost":561.9679161757083},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2-0905","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":2.52e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017203200000000002,"max_completion_cost":0.08257536,"max_cost":0.09289728},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0038761397985191764,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":26.461114357890914,"max_completion_cost":127.01334891787637,"max_cost":142.8900175326109},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-70b","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":1.365e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017891328,"max_completion_cost":0.05505024,"max_cost":0.05505024},"sats_pricing":{"prompt":0.00020995757241978874,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.51955893220655,"max_completion_cost":84.67556594525091,"max_cost":84.67556594525091},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-4-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-405b","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":3.1500000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1376256,"max_completion_cost":0.41287680000000004,"max_cost":0.41287680000000004},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.0048451747481489706,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":211.68891486312725,"max_completion_cost":635.0667445893819,"max_cost":635.0667445893819},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-4-405b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.625e-07,"completion":9.975e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.365e-07,"input_cache_write":0.0,"max_prompt_cost":0.043008,"max_completion_cost":0.03268608,"max_cost":0.06709248},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0015343053369138405,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00020995757241978874,"input_cache_write":0.0,"max_prompt_cost":66.15278589472727,"max_completion_cost":50.276117279992725,"max_cost":103.19834599577455},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3.1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2000000000000006e-08,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.2752512,"max_cost":0.2752512},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.460232997531961e-05,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":423.3778297262545,"max_cost":423.3778297262545},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5v","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":1.89e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1550000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.04128768,"max_completion_cost":0.03096576,"max_cost":0.061931520000000004},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.002907104848889382,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017765640743212894,"input_cache_write":0.0,"max_prompt_cost":63.506674458938186,"max_completion_cost":47.63000584420364,"max_cost":95.26001168840727},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5v","alias_ids":null,"forwarded_model_id":null},{"id":"jamba-large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":8.4e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5376,"max_completion_cost":0.0344064,"max_cost":0.5634047999999999},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.01292046599506392,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":826.9098236840908,"max_completion_cost":52.92222871578181,"max_cost":866.6014952209272},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"ai21/jamba-large-1.7","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-07,"input_cache_write":0.0,"max_prompt_cost":0.525,"max_completion_cost":1.344,"max_cost":1.701},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0,"max_prompt_cost":807.5291246914951,"max_completion_cost":2067.2745592102274,"max_cost":2616.394364000444},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5:batch","name":"OpenAI: GPT-5 (batch)","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.562500000000001e-07,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":6.5625e-08,"input_cache_write":0.0,"max_prompt_cost":0.2625,"max_completion_cost":0.672,"max_cost":0.8505},"sats_pricing":{"prompt":0.001009411405864369,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00010094114058643687,"input_cache_write":0.0,"max_prompt_cost":403.76456234574755,"max_completion_cost":1033.6372796051137,"max_cost":1308.197182000222},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.105,"max_completion_cost":0.2688,"max_cost":0.34019999999999995},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":161.505824938299,"max_completion_cost":413.4549118420454,"max_cost":523.2788728000887},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini:batch","name":"OpenAI: GPT-5 Mini (batch)","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.0525,"max_completion_cost":0.1344,"max_cost":0.17009999999999997},"sats_pricing":{"prompt":0.00020188228117287374,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":2.0188228117287376e-05,"input_cache_write":0.0,"max_prompt_cost":80.7529124691495,"max_completion_cost":206.7274559210227,"max_cost":261.63943640004436},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.250000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.021,"max_completion_cost":0.05376,"max_cost":0.06804},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":8.075291246914952e-06,"input_cache_write":0.0,"max_prompt_cost":32.3011649876598,"max_completion_cost":82.6909823684091,"max_cost":104.65577456001776},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano:batch","name":"OpenAI: GPT-5 Nano (batch)","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.625e-08,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.6250000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.0105,"max_completion_cost":0.02688,"max_cost":0.03402},"sats_pricing":{"prompt":4.037645623457475e-05,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":4.037645623457476e-06,"input_cache_write":0.0,"max_prompt_cost":16.1505824938299,"max_completion_cost":41.34549118420455,"max_cost":52.32788728000888},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.885e-08,"completion":1.785e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0050921472,"max_completion_cost":0.023396352,"max_cost":0.023396352},"sats_pricing":{"prompt":5.975715522717063e-05,"completion":0.0002745599023951083,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.832489849935709,"max_completion_cost":35.987115526731635,"max_cost":35.987115526731635},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-oss-120b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.15e-08,"completion":1.365e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.004128768,"max_completion_cost":0.017891328,"max_cost":0.017891328},"sats_pricing":{"prompt":4.8451747481489696e-05,"completion":0.00020995757241978874,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":6.3506674458938175,"max_completion_cost":27.51955893220655,"max_cost":27.51955893220655},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-oss-20b","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":7.874999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.5750000000000002e-06,"input_cache_write":1.9687499999999997e-05,"max_prompt_cost":3.15,"max_completion_cost":2.5199999999999996,"max_cost":5.1659999999999995},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.12112936870372425,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0024225873740744853,"input_cache_write":0.03028234217593106,"max_prompt_cost":4845.17474814897,"max_completion_cost":3876.1397985191757,"max_cost":7946.08658696431},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1:batch","name":"Anthropic: Claude Opus 4.1 (batch)","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":7.875e-06,"completion":3.9374999999999995e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":7.875000000000001e-07,"input_cache_write":9.843749999999999e-06,"max_prompt_cost":1.575,"max_completion_cost":1.2599999999999998,"max_cost":2.5829999999999997},"sats_pricing":{"prompt":0.012112936870372426,"completion":0.06056468435186212,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0012112936870372426,"input_cache_write":0.01514117108796553,"max_prompt_cost":2422.587374074485,"max_completion_cost":1938.0698992595878,"max_cost":3973.043293482155},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-2508","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":9.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.08064,"max_completion_cost":0.24192,"max_cost":0.24192},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.001453552424444691,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":124.03647355261364,"max_completion_cost":372.10942065784093,"max_cost":372.10942065784093},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/codestral-2508","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.35e-08,"completion":2.835e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01176,"max_completion_cost":0.009289728,"max_cost":0.018641280000000003},"sats_pricing":{"prompt":0.00011305407745680931,"completion":0.00043606572733340734,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.08865239308949,"max_completion_cost":14.289001753261092,"max_cost":28.67309813624586},"per_request_limits":null,"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":5.05575e-08,"completion":2.027025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00647136,"max_completion_cost":0.00648648,"max_cost":0.01134},"sats_pricing":{"prompt":7.776505470779098e-05,"completion":0.0003117869950433862,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.953927002597244,"max_completion_cost":9.97718384138836,"max_cost":17.442629093336294},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1550000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.08257536,"max_completion_cost":0.22708224000000005,"max_cost":0.24772608000000004},"sats_pricing":{"prompt":0.0009690349496297941,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017765640743212894,"input_cache_write":0.0,"max_prompt_cost":127.01334891787637,"max_completion_cost":349.2867095241601,"max_cost":381.04004675362916},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5-air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.365e-07,"completion":8.925e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.017891328,"max_completion_cost":0.08773632,"max_cost":0.092209152},"sats_pricing":{"prompt":0.00020995757241978874,"completion":0.0013727995119755417,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":27.51955893220655,"max_completion_cost":134.95168322524364,"max_cost":141.83157295829528},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-4.5-air","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.415e-07,"completion":2.415e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.031653888,"max_completion_cost":0.31653888,"max_cost":0.31653888},"sats_pricing":{"prompt":0.0003714633973580877,"completion":0.0037146339735808775,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":48.688450418519274,"max_completion_cost":486.8845041851928,"max_cost":486.8845041851928},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.08257536,"max_completion_cost":0.0688128,"max_cost":0.13074432},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":127.01334891787637,"max_completion_cost":105.84445743156363,"max_cost":201.1044691199709},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.01344,"max_completion_cost":0.00043008,"max_cost":0.01365504},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":20.672745592102274,"max_completion_cost":0.6615278589472727,"max_cost":21.00350952157591},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"bytedance/ui-tars-1.5-7b","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":4.2e-07,"request":0.0,"image":1.05e-07,"web_search":0.014700000000000001,"internal_reasoning":4.2e-07,"input_cache_read":1.0500000000000001e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.11010048,"max_completion_cost":0.0275247,"max_cost":0.130744005},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000646023299753196,"request":0.001,"image":0.000161505824938299,"web_search":22.610815491361862,"internal_reasoning":0.000646023299753196,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":169.35113189050182,"max_completion_cost":42.3371369493257,"max_cost":201.1039846024961},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite:batch","name":"Google: Gemini 2.5 Flash Lite (batch)","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":2.1e-07,"request":0.0,"image":5.25e-08,"web_search":0.014700000000000001,"internal_reasoning":2.1e-07,"input_cache_read":1.0500000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.01376235,"max_cost":0.0653720025},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000323011649876598,"request":0.001,"image":8.07529124691495e-05,"web_search":22.610815491361862,"internal_reasoning":0.000323011649876598,"input_cache_read":1.6150582493829903e-05,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":21.16856847466285,"max_cost":100.55199230124805},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.45e-08,"completion":5.775000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.024772608,"max_completion_cost":0.009461760000000001,"max_cost":0.03268608},"sats_pricing":{"prompt":0.00014535524244446913,"completion":0.0008882820371606447,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.104004675362916,"max_completion_cost":14.553612896840002,"max_cost":50.276117279992725},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.985000000000001e-07,"completion":2.415e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07844659200000001,"max_completion_cost":0.24235008000000002,"max_cost":0.260736},"sats_pricing":{"prompt":0.0009205832021483045,"completion":0.0037146339735808775,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":120.66268147198257,"max_completion_cost":372.7709485167882,"max_cost":401.05126448678413},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2","alias_ids":null,"forwarded_model_id":null},{"id":"dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":9.45e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02688,"max_completion_cost":0.00774144,"max_cost":0.03290112},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.001453552424444691,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":41.34549118420455,"max_completion_cost":11.90750146105091,"max_cost":50.606881209466366},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"venice/uncensored","alias_ids":null,"forwarded_model_id":null},{"id":"hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.47e-07,"completion":5.985000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019267584,"max_completion_cost":0.07844659200000001,"max_cost":0.07844659200000001},"sats_pricing":{"prompt":0.00022610815491361862,"completion":0.0009205832021483045,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":29.63644808083782,"max_completion_cost":120.66268147198257,"max_cost":120.66268147198257},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"tencent/hunyuan-a13b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-large","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.45e-07,"completion":1.995e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.24772608,"max_completion_cost":0.26148864,"max_cost":0.38535168},"sats_pricing":{"prompt":0.001453552424444691,"completion":0.003068610673827681,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":381.0400467536291,"max_completion_cost":402.2089382399418,"max_cost":592.7289616167564},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"morph/morph-v3-large","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-fast","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.4e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06881280000000001,"max_completion_cost":0.04788,"max_cost":0.08477280000000001},"sats_pricing":{"prompt":0.001292046599506392,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":105.84445743156365,"max_completion_cost":73.64665617186435,"max_cost":130.3933428221851},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"morph/morph-v3-fast","alias_ids":null,"forwarded_model_id":null},{"id":"ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.4100000000000004e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.054243000000000006,"max_completion_cost":0.021,"max_cost":0.06818700000000001},"sats_pricing":{"prompt":0.0006783244647408559,"completion":0.002018822811728738,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":83.43390916312528,"max_completion_cost":32.3011649876598,"max_cost":104.8818827149314},"per_request_limits":null,"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":9.843750000000001e-08,"completion":2.625e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.025200000000000004,"max_completion_cost":0.0043008,"max_cost":0.027888000000000003},"sats_pricing":{"prompt":0.00015141171087965533,"completion":0.0004037645623457475,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.76139798519177,"max_completion_cost":6.615278589472727,"max_cost":42.89594710361222},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m1","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5775000000000001,"max_completion_cost":0.09240000000000001,"max_cost":0.6468000000000002},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":888.2820371606448,"max_completion_cost":142.12512594570313,"max_cost":994.8758816199221},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m1","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.15e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":3.15e-07,"web_search":0.014700000000000001,"internal_reasoning":2.6250000000000003e-06,"input_cache_read":3.15e-08,"input_cache_write":8.749999999999997e-08,"max_prompt_cost":0.33030144,"max_completion_cost":0.172029375,"max_cost":0.48168729},"sats_pricing":{"prompt":0.00048451747481489705,"completion":0.004037645623457476,"request":0.001,"image":0.00048451747481489705,"web_search":22.610815491361862,"internal_reasoning":0.004037645623457476,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.00013458818744858247,"max_prompt_cost":508.0533956715055,"max_completion_cost":264.60710593328565,"max_cost":740.9076488927968},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash:batch","name":"Google: Gemini 2.5 Flash (batch)","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":1.575e-07,"web_search":0.014700000000000001,"internal_reasoning":1.3125000000000001e-06,"input_cache_read":3.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.16515072,"max_completion_cost":0.0860146875,"max_cost":0.240843645},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.002018822811728738,"request":0.001,"image":0.00024225873740744852,"web_search":22.610815491361862,"internal_reasoning":0.002018822811728738,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.0,"max_prompt_cost":254.02669783575274,"max_completion_cost":132.30355296664283,"max_cost":370.4538244463984},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":1.3125000000000001e-06,"web_search":0.014700000000000001,"internal_reasoning":1.0500000000000001e-05,"input_cache_read":1.3125e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":1.3762560000000001,"max_completion_cost":0.6881280000000001,"max_cost":1.9783680000000001},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.002018822811728738,"web_search":22.610815491361862,"internal_reasoning":0.016150582493829904,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0006056468435186213,"max_prompt_cost":2116.889148631273,"max_completion_cost":1058.4445743156366,"max_cost":3043.028151157455},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro:batch","name":"Google: Gemini 2.5 Pro (batch)","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":6.562500000000001e-07,"completion":5.2500000000000006e-06,"request":0.0,"image":6.562500000000001e-07,"web_search":0.014700000000000001,"internal_reasoning":5.2500000000000006e-06,"input_cache_read":1.3125e-07,"input_cache_write":0.0,"max_prompt_cost":0.6881280000000001,"max_completion_cost":0.34406400000000004,"max_cost":0.9891840000000001},"sats_pricing":{"prompt":0.001009411405864369,"completion":0.008075291246914952,"request":0.001,"image":0.001009411405864369,"web_search":22.610815491361862,"internal_reasoning":0.008075291246914952,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0,"max_prompt_cost":1058.4445743156366,"max_completion_cost":529.2222871578183,"max_cost":1521.5140755787274},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1000000000000002e-05,"completion":8.400000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.2,"max_completion_cost":8.4,"max_cost":10.5},"sats_pricing":{"prompt":0.03230116498765981,"completion":0.12920465995063923,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6460.232997531961,"max_completion_cost":12920.465995063922,"max_cost":16150.582493829901},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro:batch","name":"OpenAI: o3 Pro (batch)","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":4.2000000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.1,"max_completion_cost":4.2,"max_cost":5.25},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.06460232997531962,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3230.1164987659804,"max_completion_cost":6460.232997531961,"max_cost":8075.2912469149505},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":1.3125000000000001e-06,"web_search":0.014700000000000001,"internal_reasoning":1.0500000000000001e-05,"input_cache_read":1.3125e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":1.3762560000000001,"max_completion_cost":0.6881280000000001,"max_cost":1.9783680000000001},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.002018822811728738,"web_search":22.610815491361862,"internal_reasoning":0.016150582493829904,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0006056468435186213,"max_prompt_cost":2116.889148631273,"max_completion_cost":1058.4445743156366,"max_cost":3043.028151157455},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro-preview-06-05","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":5.25e-07,"completion":2.2575000000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.675e-07,"input_cache_write":0.0,"max_prompt_cost":0.086016,"max_completion_cost":0.07397376000000001,"max_cost":0.14278656},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0034723752361734295,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005652703872840465,"input_cache_write":0.0,"max_prompt_cost":132.30557178945455,"max_completion_cost":113.78279173893094,"max_cost":219.62724917049457},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1-0528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":7.874999999999999e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.5750000000000002e-06,"input_cache_write":1.9687499999999997e-05,"max_prompt_cost":3.15,"max_completion_cost":2.5199999999999996,"max_cost":5.1659999999999995},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.12112936870372425,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0024225873740744853,"input_cache_write":0.03028234217593106,"max_prompt_cost":4845.17474814897,"max_completion_cost":3876.1397985191757,"max_cost":7946.08658696431},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4-opus-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.15e-07,"input_cache_write":3.9375e-06,"max_prompt_cost":0.6300000000000001,"max_completion_cost":1.008,"max_cost":1.4364000000000001},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00048451747481489705,"input_cache_write":0.006056468435186213,"max_prompt_cost":969.0349496297943,"max_completion_cost":1550.4559194076705,"max_cost":2209.3996851559305},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4-sonnet-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-08,"completion":1.26e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002064384,"max_completion_cost":0.004128768,"max_cost":0.004128768},"sats_pricing":{"prompt":9.690349496297939e-05,"completion":0.00019380698992595879,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.1753337229469087,"max_completion_cost":6.3506674458938175,"max_cost":6.3506674458938175},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3n-e4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2000000000000006e-08,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.2752512,"max_cost":0.2752512},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.460232997531961e-05,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":423.3778297262545,"max_cost":423.3778297262545},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-medium-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":1.3125000000000001e-06,"web_search":0.014700000000000001,"internal_reasoning":1.0500000000000001e-05,"input_cache_read":1.3125e-07,"input_cache_write":3.9375000000000004e-07,"max_prompt_cost":1.3762560000000001,"max_completion_cost":0.6881175,"max_cost":1.9783588125000002},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.016150582493829904,"request":0.001,"image":0.002018822811728738,"web_search":22.610815491361862,"internal_reasoning":0.016150582493829904,"input_cache_read":0.00020188228117287374,"input_cache_write":0.0006056468435186213,"max_prompt_cost":2116.889148631273,"max_completion_cost":1058.4284237331426,"max_cost":3043.0140193977727},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro-preview-03-25","alias_ids":null,"forwarded_model_id":null},{"id":"virtuoso-large","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.875000000000001e-07,"completion":1.26e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.10321920000000001,"max_completion_cost":0.08064,"max_cost":0.1334592},"sats_pricing":{"prompt":0.0012112936870372426,"completion":0.0019380698992595882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":158.76668614734547,"max_completion_cost":124.03647355261364,"max_cost":205.28036372957558},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"arcee-ai/virtuoso-large","alias_ids":null,"forwarded_model_id":null},{"id":"llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.89e-07,"completion":1.89e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030965760000000002,"max_completion_cost":0.003096576,"max_cost":0.030965760000000002},"sats_pricing":{"prompt":0.00029071048488893826,"completion":0.00029071048488893826,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":47.63000584420364,"max_completion_cost":4.7630005844203644,"max_cost":47.63000584420364},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-guard-4-12b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.26e-07,"completion":5.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005160959999999999,"max_completion_cost":0.0086016,"max_cost":0.011698176},"sats_pricing":{"prompt":0.00019380698992595879,"completion":0.000807529124691495,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.938334307367271,"max_completion_cost":13.230557178945453,"max_cost":17.993557763365818},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-30b-a3b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-8b","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.2285000000000002e-07,"completion":4.7775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.016102195200000002,"max_completion_cost":0.003913728,"max_cost":0.019009536},"sats_pricing":{"prompt":0.00018896181517780988,"completion":0.0007348515034692605,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24.767603038985897,"max_completion_cost":6.019903516420182,"max_cost":29.239531365469457},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-8b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-14b","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.38875e-07,"completion":9.555e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.031309824,"max_completion_cost":0.007827456,"max_cost":0.037180416},"sats_pricing":{"prompt":0.0003674257517346303,"completion":0.001469703006938521,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":48.15922813136146,"max_completion_cost":12.039807032840365,"max_cost":57.18908340599173},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-14b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-32b","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":8.400000000000001e-08,"completion":2.94e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0034406400000000005,"max_completion_cost":0.004816896,"max_cost":0.00688128},"sats_pricing":{"prompt":0.00012920465995063923,"completion":0.00045221630982723724,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.292222871578183,"max_completion_cost":7.409112020209455,"max_cost":10.584445743156364},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-32b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":4.7775e-07,"completion":1.911e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.062619648,"max_completion_cost":0.015654912,"max_cost":0.074360832},"sats_pricing":{"prompt":0.0007348515034692605,"completion":0.002939406013877042,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":96.31845626272292,"max_completion_cost":24.07961406568073,"max_cost":114.37816681198346},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-235b-a22b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":4.620000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.8875000000000004e-07,"input_cache_write":0.0,"max_prompt_cost":0.23100000000000004,"max_completion_cost":0.4620000000000001,"max_cost":0.5775000000000001},"sats_pricing":{"prompt":0.0017765640743212894,"completion":0.0071062562972851575,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00044414101858032234,"input_cache_write":0.0,"max_prompt_cost":355.31281486425786,"max_completion_cost":710.6256297285157,"max_cost":888.2820371606448},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high:batch","name":"OpenAI: o4 Mini High (batch)","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.4437500000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.11550000000000002,"max_completion_cost":0.23100000000000004,"max_cost":0.28875000000000006},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00022207050929016117,"input_cache_write":0.0,"max_prompt_cost":177.65640743212893,"max_completion_cost":355.31281486425786,"max_cost":444.1410185803224},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":8.4e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":0.0,"max_prompt_cost":0.42,"max_completion_cost":0.84,"max_cost":1.05},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.01292046599506392,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.0,"max_prompt_cost":646.023299753196,"max_completion_cost":1292.046599506392,"max_cost":1615.0582493829902},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3:batch","name":"OpenAI: o3 (batch)","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.21,"max_completion_cost":0.42,"max_cost":0.525},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":323.011649876598,"max_completion_cost":646.023299753196,"max_cost":807.5291246914951},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":4.620000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.8875000000000004e-07,"input_cache_write":0.0,"max_prompt_cost":0.23100000000000004,"max_completion_cost":0.4620000000000001,"max_cost":0.5775000000000001},"sats_pricing":{"prompt":0.0017765640743212894,"completion":0.0071062562972851575,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00044414101858032234,"input_cache_write":0.0,"max_prompt_cost":355.31281486425786,"max_completion_cost":710.6256297285157,"max_cost":888.2820371606448},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini:batch","name":"OpenAI: o4 Mini (batch)","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.4437500000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.11550000000000002,"max_completion_cost":0.23100000000000004,"max_cost":0.28875000000000006},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00022207050929016117,"input_cache_write":0.0,"max_prompt_cost":177.65640743212893,"max_completion_cost":355.31281486425786,"max_cost":444.1410185803224},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":8.4e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-07,"input_cache_write":0.0,"max_prompt_cost":2.1999096,"max_completion_cost":0.2752512,"max_cost":2.406348},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.01292046599506392,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000807529124691495,"input_cache_write":0.0,"max_prompt_cost":3383.7925213112703,"max_completion_cost":423.3778297262545,"max_cost":3701.3258936059615},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1:batch","name":"OpenAI: GPT-4.1 (batch)","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":1.0999548,"max_completion_cost":0.1376256,"max_cost":1.203174},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":1691.8962606556352,"max_completion_cost":211.68891486312725,"max_cost":1850.6629468029807},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.2e-07,"completion":1.68e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.43998192,"max_completion_cost":0.05505024,"max_cost":0.48126959999999996},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.002584093199012784,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.000161505824938299,"input_cache_write":0.0,"max_prompt_cost":676.7585042622542,"max_completion_cost":84.67556594525091,"max_cost":740.2651787211922},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini:batch","name":"OpenAI: GPT-4.1 Mini (batch)","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":8.4e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.21999096,"max_completion_cost":0.02752512,"max_cost":0.24063479999999998},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.001292046599506392,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":8.07529124691495e-05,"input_cache_write":0.0,"max_prompt_cost":338.3792521311271,"max_completion_cost":42.337782972625455,"max_cost":370.1325893605961},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.10999548,"max_completion_cost":0.01376256,"max_cost":0.12031739999999999},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":169.18962606556354,"max_completion_cost":21.168891486312727,"max_cost":185.06629468029806},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano:batch","name":"OpenAI: GPT-4.1 Nano (batch)","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":1.3125e-08,"input_cache_write":0.0,"max_prompt_cost":0.05499774,"max_completion_cost":0.00688128,"max_cost":0.060158699999999996},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":2.0188228117287376e-05,"input_cache_write":0.0,"max_prompt_cost":84.59481303278177,"max_completion_cost":10.584445743156364,"max_cost":92.53314734014903},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-maverick","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":8.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.22020096,"max_completion_cost":0.01376256,"max_cost":0.23052288},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.001292046599506392,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":338.70226378100364,"max_completion_cost":21.168891486312727,"max_cost":354.5789323957382},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-scout","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.034406400000000004,"max_completion_cost":0.00516096,"max_cost":0.03784704},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":52.92222871578183,"max_completion_cost":7.938334307367273,"max_cost":58.21445158736},"per_request_limits":null,"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.835e-07,"completion":1.176e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4175e-07,"input_cache_write":0.0,"max_prompt_cost":0.04644864,"max_completion_cost":0.077070336,"max_cost":0.10493952000000001},"sats_pricing":{"prompt":0.00043606572733340734,"completion":0.001808865239308949,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021803286366670367,"input_cache_write":0.0,"max_prompt_cost":71.44500876630545,"max_completion_cost":118.54579232335128,"max_cost":161.41279758313456},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3-0324","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":0.00015749999999999998,"completion":0.0006299999999999999,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":31.499999999999996,"max_completion_cost":62.99999999999999,"max_cost":78.74999999999999},"sats_pricing":{"prompt":0.2422587374074485,"completion":0.969034949629794,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":48451.7474814897,"max_completion_cost":96903.4949629794,"max_cost":121129.36870372423},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro:batch","name":"OpenAI: o1-pro (batch)","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.874999999999999e-05,"completion":0.00031499999999999996,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":15.749999999999998,"max_completion_cost":31.499999999999996,"max_cost":39.37499999999999},"sats_pricing":{"prompt":0.12112936870372425,"completion":0.484517474814897,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24225.87374074485,"max_completion_cost":48451.7474814897,"max_cost":60564.68435186212},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.6855e-07,"completion":5.827500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0471744,"max_completion_cost":0.074592,"max_cost":0.074592},"sats_pricing":{"prompt":0.0005668854455334295,"completion":0.0008963573284075596,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.56133702827897,"max_completion_cost":114.73373803616762,"max_cost":114.73373803616762},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.25e-08,"completion":1.05e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.00172032,"max_cost":0.00774144},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.000161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":2.646111435789091,"max_cost":11.90750146105091},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.25e-08,"completion":1.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.00258048,"max_cost":0.0086016},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.00024225873740744852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":3.9691671536836366,"max_cost":13.230557178945453},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-12b-it","alias_ids":null,"forwarded_model_id":null},{"id":"command-a","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.672,"max_completion_cost":0.08601600000000001,"max_cost":0.7365120000000001},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1033.6372796051137,"max_completion_cost":132.30557178945458,"max_cost":1132.8664584472046},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-a-03-2025","alias_ids":null,"forwarded_model_id":null},{"id":"reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.01376256,"max_cost":0.01376256},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":21.168891486312727,"max_cost":21.168891486312727},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"rekaai/reka-flash-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":8.400000000000001e-08,"completion":4.725e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2000000000000006e-08,"input_cache_write":0.0,"max_prompt_cost":0.011010048000000001,"max_completion_cost":0.06193152,"max_cost":0.06193152},"sats_pricing":{"prompt":0.00012920465995063923,"completion":0.0007267762122223455,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.460232997531961e-05,"input_cache_write":0.0,"max_prompt_cost":16.935113189050185,"max_completion_cost":95.26001168840727,"max_cost":95.26001168840727},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-3-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":8.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-07,"input_cache_write":0.0,"max_prompt_cost":0.018923520000000003,"max_completion_cost":0.02752512,"max_cost":0.02752512},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.001292046599506392,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004037645623457475,"input_cache_write":0.0,"max_prompt_cost":29.107225793680005,"max_completion_cost":42.337782972625455,"max_cost":42.337782972625455},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/skyfall-36b-v2","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.1e-06,"completion":8.4e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":1.0752,"max_cost":1.0752},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.01292046599506392,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":413.4549118420454,"max_completion_cost":1653.8196473681817,"max_cost":1653.8196473681817},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-reasoning-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6300000000000001,"max_completion_cost":0.126,"max_cost":0.7308000000000001},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":969.0349496297943,"max_completion_cost":193.80698992595882,"max_cost":1124.0805415705613},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-deep-research","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.1e-06,"completion":8.4e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":3.1500000000000003e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":1.0752,"max_cost":1.0752},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.01292046599506392,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0048451747481489706,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":413.4549118420454,"max_completion_cost":1653.8196473681817,"max_cost":1653.8196473681817},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar-deep-research","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-saba","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1000000000000003e-08,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.02064384,"max_cost":0.02064384},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.230116498765981e-05,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":31.753337229469093,"max_cost":31.753337229469093},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-saba-2502","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":4.620000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.775000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.23100000000000004,"max_completion_cost":0.4620000000000001,"max_cost":0.5775000000000001},"sats_pricing":{"prompt":0.0017765640743212894,"completion":0.0071062562972851575,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0008882820371606447,"input_cache_write":0.0,"max_prompt_cost":355.31281486425786,"max_completion_cost":710.6256297285157,"max_cost":888.2820371606448},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high:batch","name":"OpenAI: o3 Mini High (batch)","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.8875000000000004e-07,"input_cache_write":0.0,"max_prompt_cost":0.11550000000000002,"max_completion_cost":0.23100000000000004,"max_cost":0.28875000000000006},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00044414101858032234,"input_cache_write":0.0,"max_prompt_cost":177.65640743212893,"max_completion_cost":355.31281486425786,"max_cost":444.1410185803224},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.4e-07,"completion":1.68e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02752512,"max_completion_cost":0.05505024,"max_cost":0.05505024},"sats_pricing":{"prompt":0.001292046599506392,"completion":0.002584093199012784,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.337782972625455,"max_completion_cost":84.67556594525091,"max_cost":84.67556594525091},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":7.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0084,"max_completion_cost":0.025200000000000004,"max_cost":0.025200000000000004},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0012112936870372426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":12.92046599506392,"max_completion_cost":38.76139798519177,"max_cost":38.76139798519177},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.73e-07,"completion":8.190000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.4600000000000006e-08,"input_cache_write":3.4125e-07,"max_prompt_cost":0.273,"max_completion_cost":0.026836992000000004,"max_cost":0.290891328},"sats_pricing":{"prompt":0.0004199151448395775,"completion":0.0012597454345187325,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.39830289679155e-05,"input_cache_write":0.0005248939310494718,"max_prompt_cost":419.91514483957747,"max_completion_cost":41.279338398309825,"max_cost":447.434703771784},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-plus-2025-01-25","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":4.620000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":5.775000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.23100000000000004,"max_completion_cost":0.4620000000000001,"max_cost":0.5775000000000001},"sats_pricing":{"prompt":0.0017765640743212894,"completion":0.0071062562972851575,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0008882820371606447,"input_cache_write":0.0,"max_prompt_cost":355.31281486425786,"max_completion_cost":710.6256297285157,"max_cost":888.2820371606448},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini:batch","name":"OpenAI: o3 Mini (batch)","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.775000000000001e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":2.8875000000000004e-07,"input_cache_write":0.0,"max_prompt_cost":0.11550000000000002,"max_completion_cost":0.23100000000000004,"max_cost":0.28875000000000006},"sats_pricing":{"prompt":0.0008882820371606447,"completion":0.0035531281486425787,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.00044414101858032234,"input_cache_write":0.0,"max_prompt_cost":177.65640743212893,"max_completion_cost":355.31281486425786,"max_cost":444.1410185803224},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.25e-08,"completion":8.400000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00172032,"max_completion_cost":0.0013762560000000002,"max_cost":0.002236416},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.00012920465995063923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.646111435789091,"max_completion_cost":2.116889148631273,"max_cost":3.4399448665258188},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","alias_ids":null,"forwarded_model_id":null},{"id":"sonar","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.00525,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.13342559999999998,"max_completion_cost":0.13342559999999998,"max_cost":0.13342559999999998},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":8.07529124691495,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":205.2286818655953,"max_completion_cost":205.2286818655953,"max_cost":205.2286818655953},"per_request_limits":null,"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/sonar","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":8.4e-07,"completion":8.4e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.00688128,"max_cost":0.00688128},"sats_pricing":{"prompt":0.001292046599506392,"completion":0.001292046599506392,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":10.584445743156364,"max_cost":10.584445743156364},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":7.35e-07,"completion":2.6250000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04704,"max_completion_cost":0.042,"max_cost":0.07728},"sats_pricing":{"prompt":0.001130540774568093,"completion":0.004037645623457476,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.35460957235796,"max_completion_cost":64.6023299753196,"max_cost":118.86828715458807},"per_request_limits":null,"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-r1","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":1.1550000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.21004032,"max_completion_cost":1.15522176,"max_cost":1.15522176},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0017765640743212894,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":323.0736681133743,"max_completion_cost":1776.9051746235589,"max_cost":1776.9051746235589},"per_request_limits":null,"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-01","alias_ids":null,"forwarded_model_id":null},{"id":"phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.35e-08,"completion":1.47e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001204224,"max_completion_cost":0.002408448,"max_cost":0.002408448},"sats_pricing":{"prompt":0.00011305407745680931,"completion":0.00022610815491361862,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.8522780050523637,"max_completion_cost":3.7045560101047275,"max_cost":3.7045560101047275},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"microsoft/phi-4","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.7027000000000005e-07,"completion":1.080135e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.9400000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.03459456,"max_completion_cost":0.017282159999999998,"max_cost":0.04755240000000001},"sats_pricing":{"prompt":0.00041571599339118175,"completion":0.0016614104211402818,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.5221630982723725e-05,"input_cache_write":0.0,"max_prompt_cost":53.21164715407126,"max_completion_cost":26.58256673824451,"max_cost":73.14275799805687},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"deepseek/deepseek-chat-v3","alias_ids":null,"forwarded_model_id":null},{"id":"l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":6.825e-07,"completion":7.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08945664,"max_completion_cost":0.012902400000000001,"max_cost":0.09117696},"sats_pricing":{"prompt":0.0010497878620989436,"completion":0.0012112936870372426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":137.59779466103274,"max_completion_cost":19.845835768418183,"max_cost":140.24390609682183},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","alias_ids":null,"forwarded_model_id":null},{"id":"o1","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-05,"completion":6.3e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":7.875e-06,"input_cache_write":0.0,"max_prompt_cost":3.15,"max_completion_cost":6.3,"max_cost":7.875},"sats_pricing":{"prompt":0.024225873740744853,"completion":0.09690349496297941,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.012112936870372426,"input_cache_write":0.0,"max_prompt_cost":4845.17474814897,"max_completion_cost":9690.34949629794,"max_cost":12112.936870372427},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"o1:batch","name":"OpenAI: o1 (batch)","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-06,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.9375e-06,"input_cache_write":0.0,"max_prompt_cost":1.575,"max_completion_cost":3.15,"max_cost":3.9375},"sats_pricing":{"prompt":0.012112936870372426,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.006056468435186213,"input_cache_write":0.0,"max_prompt_cost":2422.587374074485,"max_completion_cost":4845.17474814897,"max_cost":6056.468435186213},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":3.9375e-08,"completion":1.575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00504,"max_completion_cost":0.00063,"max_cost":0.0055125},"sats_pricing":{"prompt":6.056468435186213e-05,"completion":0.00024225873740744852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.752279597038353,"max_completion_cost":0.9690349496297941,"max_cost":8.479055809260698},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r7b-12-2024","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":1.05e-07,"completion":3.3600000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01376256,"max_completion_cost":0.005505024000000001,"max_cost":0.017547264},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0005168186398025569,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.168891486312727,"max_completion_cost":8.467556594525092,"max_cost":26.99033664504873},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.3-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"nova-lite-v1","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":6.3e-08,"completion":2.52e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0189,"max_completion_cost":0.0012902399999999998,"max_cost":0.01986768},"sats_pricing":{"prompt":9.690349496297939e-05,"completion":0.00038761397985191757,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":29.071048488893823,"max_completion_cost":1.9845835768418179,"max_cost":30.559486171525183},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-micro-v1","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.675e-08,"completion":1.47e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004704,"max_completion_cost":0.0007526400000000001,"max_cost":0.00526848},"sats_pricing":{"prompt":5.6527038728404655e-05,"completion":0.00022610815491361862,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.235460957235795,"max_completion_cost":1.1576737531577275,"max_cost":8.10371627210409},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-micro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-pro-v1","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":8.4e-07,"completion":3.36e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.252,"max_completion_cost":0.017203200000000002,"max_cost":0.2649024},"sats_pricing":{"prompt":0.001292046599506392,"completion":0.005168186398025568,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":387.61397985191763,"max_completion_cost":26.461114357890914,"max_cost":407.4598156203358},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"amazon/nova-pro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3125000000000001e-06,"input_cache_write":0.0,"max_prompt_cost":0.336,"max_completion_cost":0.17203200000000002,"max_cost":0.46502400000000005},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002018822811728738,"input_cache_write":0.0,"max_prompt_cost":516.8186398025568,"max_completion_cost":264.61114357890915,"max_cost":715.2769974867388},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-2024-11-20","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2407","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.2752512,"max_completion_cost":0.8257536000000001,"max_cost":0.8257536000000001},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":423.3778297262545,"max_completion_cost":1270.1334891787637,"max_cost":1270.1334891787637},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large-2407","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":6.930000000000001e-07,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.022708224000000003,"max_completion_cost":0.0344064,"max_cost":0.0344064},"sats_pricing":{"prompt":0.0010659384445927736,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.928670952416006,"max_completion_cost":52.92222871578181,"max_cost":52.92222871578181},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":4.2e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.43008,"max_completion_cost":0.43008,"max_cost":0.43008},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":661.5278589472728,"max_completion_cost":661.5278589472728,"max_cost":661.5278589472728},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":1024000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/unslopnemo-12b","alias_ids":null,"forwarded_model_id":null},{"id":"magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.1500000000000003e-06,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051609600000000005,"max_completion_cost":0.010752000000000001,"max_cost":0.0559104},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":79.38334307367273,"max_completion_cost":16.538196473681822,"max_cost":85.99862166314546},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthracite-org/magnum-v4-72b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.05e-07,"completion":2.1e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00344064,"max_completion_cost":0.00688128,"max_cost":0.00688128},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.000323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.292222871578182,"max_completion_cost":10.584445743156364,"max_cost":10.584445743156364},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-7b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"rocinante-12b","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":2.625e-07,"completion":5.25e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0172032,"max_completion_cost":0.0344064,"max_cost":0.0344064},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.000807529124691495,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":26.461114357890906,"max_completion_cost":52.92222871578181,"max_cost":52.92222871578181},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thedrummer/rocinante-12b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.8350000000000002e-08,"completion":2.1105000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001701,"max_completion_cost":0.012663000000000002,"max_cost":0.012663000000000002},"sats_pricing":{"prompt":4.3606572733340736e-05,"completion":0.00032462670812598103,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.616394364000444,"max_completion_cost":19.477602487558865,"max_cost":19.477602487558865},"per_request_limits":null,"top_provider":{"context_length":60000,"max_completion_tokens":60000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.2-1b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.25e-08,"completion":3.4650000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.045416448000000005,"max_cost":0.045416448000000005},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.0005329692222963868,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":69.85734190483201,"max_cost":69.85734190483201},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.2-3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.78e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012386304,"max_completion_cost":0.00688128,"max_cost":0.013074432},"sats_pricing":{"prompt":0.0005814209697778765,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.052002337681458,"max_completion_cost":10.584445743156364,"max_cost":20.11044691199709},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen-2.5-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-08-2024","name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02016,"max_completion_cost":0.00252,"max_cost":0.02205},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":31.00911838815341,"max_completion_cost":3.8761397985191763,"max_cost":33.91622323704279},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.336,"max_completion_cost":0.042,"max_cost":0.3675},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":516.8186398025568,"max_completion_cost":64.6023299753196,"max_cost":565.2703872840465},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"cohere/command-r-plus-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":8.925e-07,"completion":8.925e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11698176,"max_completion_cost":0.01462272,"max_cost":0.11698176},"sats_pricing":{"prompt":0.0013727995119755417,"completion":0.0013727995119755417,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":179.9355776336582,"max_completion_cost":22.491947204207275,"max_cost":179.9355776336582},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3.1-euryale-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":7.35e-07,"completion":7.35e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09633792,"max_completion_cost":0.01204224,"max_cost":0.09633792},"sats_pricing":{"prompt":0.001130540774568093,"completion":0.001130540774568093,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":148.1822404041891,"max_completion_cost":18.522780050523636,"max_cost":148.1822404041891},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":1.05e-06,"completion":1.05e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1376256,"max_completion_cost":0.0172032,"max_cost":0.1376256},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00161505824938299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":211.68891486312725,"max_completion_cost":26.461114357890906,"max_cost":211.68891486312725},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","alias_ids":null,"forwarded_model_id":null},{"id":"l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.2000000000000006e-08,"completion":5.25e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00034406400000000005,"max_completion_cost":0.00043008,"max_cost":0.00043008},"sats_pricing":{"prompt":6.460232997531961e-05,"completion":8.07529124691495e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5292222871578183,"max_completion_cost":0.6615278589472727,"max_cost":0.6615278589472727},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sao10k/l3-lunaris-8b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3125000000000001e-06,"input_cache_write":0.0,"max_prompt_cost":0.336,"max_completion_cost":0.17203200000000002,"max_cost":0.46502400000000005},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002018822811728738,"input_cache_write":0.0,"max_prompt_cost":516.8186398025568,"max_completion_cost":264.61114357890915,"max_cost":715.2769974867388},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-2024-08-06","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.2e-07,"completion":4.2e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05505024,"max_completion_cost":0.00688128,"max_cost":0.05505024},"sats_pricing":{"prompt":0.000646023299753196,"completion":0.000646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":84.67556594525091,"max_completion_cost":10.584445743156364,"max_cost":84.67556594525091},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.1-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.25e-08,"completion":8.400000000000001e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.00688128,"max_completion_cost":0.011010048000000001,"max_cost":0.011010048000000001},"sats_pricing":{"prompt":8.07529124691495e-05,"completion":0.00012920465995063923,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.037645623457475e-05,"input_cache_write":0.0,"max_prompt_cost":10.584445743156364,"max_completion_cost":16.935113189050185,"max_cost":16.935113189050185},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"meta-llama/llama-3.1-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-nemo","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":1.9950000000000003e-08,"completion":3.15e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0026148864000000004,"max_completion_cost":0.000516096,"max_cost":0.0028041216},"sats_pricing":{"prompt":3.068610673827682e-05,"completion":4.8451747481489696e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.022089382399419,"max_completion_cost":0.7938334307367272,"max_cost":4.313161640336218},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-nemo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini","name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.02016,"max_completion_cost":0.01032192,"max_cost":0.02790144},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012112936870372426,"input_cache_write":0.0,"max_prompt_cost":31.00911838815341,"max_completion_cost":15.876668614734546,"max_cost":42.91661984920432},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":6.3e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.875e-08,"input_cache_write":0.0,"max_prompt_cost":0.02016,"max_completion_cost":0.01032192,"max_cost":0.02790144},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012112936870372426,"input_cache_write":0.0,"max_prompt_cost":31.00911838815341,"max_completion_cost":15.876668614734546,"max_cost":42.91661984920432},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-mini-2024-07-18","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini:batch","name":"OpenAI: GPT-4o-mini (batch)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-08,"completion":3.15e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.9375e-08,"input_cache_write":0.0,"max_prompt_cost":0.01008,"max_completion_cost":0.00516096,"max_cost":0.01395072},"sats_pricing":{"prompt":0.00012112936870372426,"completion":0.00048451747481489705,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":6.056468435186213e-05,"input_cache_write":0.0,"max_prompt_cost":15.504559194076705,"max_completion_cost":7.938334307367273,"max_cost":21.45830992460216},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":6.825e-07,"completion":6.825e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00559104,"max_completion_cost":0.00139776,"max_cost":0.00559104},"sats_pricing":{"prompt":0.0010497878620989436,"completion":0.0010497878620989436,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.599862166314546,"max_completion_cost":2.1499655415786365,"max_cost":8.599862166314546},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemma-2-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o","name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.6250000000000003e-06,"completion":1.0500000000000001e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3125000000000001e-06,"input_cache_write":0.0,"max_prompt_cost":0.336,"max_completion_cost":0.17203200000000002,"max_cost":0.46502400000000005},"sats_pricing":{"prompt":0.004037645623457476,"completion":0.016150582493829904,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.002018822811728738,"input_cache_write":0.0,"max_prompt_cost":516.8186398025568,"max_completion_cost":264.61114357890915,"max_cost":715.2769974867388},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.672,"max_completion_cost":0.064512,"max_cost":0.7150080000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1033.6372796051137,"max_completion_cost":99.22917884209092,"max_cost":1099.7900654998411},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-2024-05-13","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o:batch","name":"OpenAI: GPT-4o (batch)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3125000000000001e-06,"completion":5.2500000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":6.562500000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.168,"max_completion_cost":0.08601600000000001,"max_cost":0.23251200000000002},"sats_pricing":{"prompt":0.002018822811728738,"completion":0.008075291246914952,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.001009411405864369,"input_cache_write":0.0,"max_prompt_cost":258.4093199012784,"max_completion_cost":132.30557178945458,"max_cost":357.6384987433694},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.1376256,"max_completion_cost":0.41287680000000004,"max_cost":0.41287680000000004},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":211.68891486312725,"max_completion_cost":635.0667445893819,"max_cost":635.0667445893819},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mixtral-8x22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"wizardlm-2-8x22b","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":6.51e-07,"completion":6.51e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.042663285,"max_completion_cost":0.005208,"max_cost":0.042663284999999995},"sats_pricing":{"prompt":0.0010013361146174538,"completion":0.0010013361146174538,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":65.62256227145484,"max_completion_cost":8.01068891693963,"max_cost":65.62256227145483},"per_request_limits":null,"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"microsoft/wizardlm-2-8x22b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.344,"max_completion_cost":0.129024,"max_cost":1.4300160000000002},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2067.2745592102274,"max_completion_cost":198.45835768418183,"max_cost":2199.5801309996823},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo:batch","name":"OpenAI: GPT-4 Turbo (batch)","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.2500000000000006e-06,"completion":1.575e-05,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.672,"max_completion_cost":0.064512,"max_cost":0.7150080000000001},"sats_pricing":{"prompt":0.008075291246914952,"completion":0.024225873740744853,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1033.6372796051137,"max_completion_cost":99.22917884209092,"max_cost":1099.7900654998411},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"claude-3-haiku","name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":1.3125000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":3.15e-08,"input_cache_write":3.15e-07,"max_prompt_cost":0.0525,"max_completion_cost":0.005376000000000001,"max_cost":0.05680079999999999},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.002018822811728738,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":4.8451747481489696e-05,"input_cache_write":0.00048451747481489705,"max_prompt_cost":80.7529124691495,"max_completion_cost":8.269098236840911,"max_cost":87.36819105862222},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-3-haiku","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large","name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1e-06,"completion":6.300000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":0.8064000000000001,"max_cost":0.8064000000000001},"sats_pricing":{"prompt":0.00323011649876598,"completion":0.009690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000323011649876598,"input_cache_write":0.0,"max_prompt_cost":413.4549118420454,"max_completion_cost":1240.3647355261367,"max_cost":1240.3647355261367},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-large","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.05e-06,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00429975,"max_completion_cost":0.0085995,"max_cost":0.0085995},"sats_pricing":{"prompt":0.00161505824938299,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.613663531223344,"max_completion_cost":13.227327062446689,"max_cost":13.227327062446689},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo-0613","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-05,"completion":3.15e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.344,"max_completion_cost":0.129024,"max_cost":1.4300160000000002},"sats_pricing":{"prompt":0.016150582493829904,"completion":0.048451747481489706,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2067.2745592102274,"max_completion_cost":198.45835768418183,"max_cost":2199.5801309996823},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4-turbo-preview","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":1.5750000000000002e-06,"completion":2.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0064496250000000005,"max_completion_cost":0.0085995,"max_cost":0.0085995},"sats_pricing":{"prompt":0.0024225873740744853,"completion":0.00323011649876598,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.920495296835018,"max_completion_cost":13.227327062446689,"max_cost":13.227327062446689},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.1500000000000003e-06,"completion":4.2e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051612750000000006,"max_completion_cost":0.0172032,"max_cost":0.055913550000000006},"sats_pricing":{"prompt":0.0048451747481489706,"completion":0.00646023299753196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":79.38818824842089,"max_completion_cost":26.461114357890906,"max_cost":86.00346683789361},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo-16k","alias_ids":null,"forwarded_model_id":null},{"id":"weaver","name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":5.25e-07,"completion":7.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0042,"max_completion_cost":0.004725000000000001,"max_cost":0.005775000000000001},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0012112936870372426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.46023299753196,"max_completion_cost":7.2677621222234565,"max_cost":8.882820371606446},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":6000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mancer/weaver","alias_ids":null,"forwarded_model_id":null},{"id":"remm-slerp-l2-13b","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":4.725e-07,"completion":6.825e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00290304,"max_completion_cost":0.004193280000000001,"max_cost":0.004193280000000001},"sats_pricing":{"prompt":0.0007267762122223455,"completion":0.0010497878620989436,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.465313047894091,"max_completion_cost":6.44989662473591,"max_cost":6.44989662473591},"per_request_limits":null,"top_provider":{"context_length":6144,"max_completion_tokens":6144,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"undi95/remm-slerp-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"mythomax-l2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":8.400000000000001e-08,"completion":1.1550000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00034406400000000005,"max_completion_cost":0.00047308800000000003,"max_cost":0.00047308800000000003},"sats_pricing":{"prompt":0.00012920465995063923,"completion":0.00017765640743212894,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5292222871578183,"max_completion_cost":0.7276806448420001,"max_cost":0.7276806448420001},"per_request_limits":null,"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"gryphe/mythomax-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo","name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":1.5750000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.008602124999999999,"max_completion_cost":0.006451200000000001,"max_cost":0.012902924999999999},"sats_pricing":{"prompt":0.000807529124691495,"completion":0.0024225873740744853,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.231364708070144,"max_completion_cost":9.922917884209092,"max_cost":19.846643297542872},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo:batch","name":"OpenAI: GPT-3.5 Turbo (batch)","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.625e-07,"completion":7.875000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0105,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004301062499999999,"max_completion_cost":0.0032256000000000003,"max_cost":0.0064514624999999996},"sats_pricing":{"prompt":0.0004037645623457475,"completion":0.0012112936870372426,"request":0.001,"image":0.0,"web_search":16.1505824938299,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.615682354035072,"max_completion_cost":4.961458942104546,"max_cost":9.923321648771436},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4","name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.15e-05,"completion":6.3e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2580165,"max_completion_cost":0.258048,"max_cost":0.3870405},"sats_pricing":{"prompt":0.048451747481489706,"completion":0.09690349496297941,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":396.86826362088215,"max_completion_cost":396.91671536836367,"max_cost":595.326621305064},"per_request_limits":null,"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-multimodal-3.5","name":"VoyageAI by MongoDB: voyage-multimodal-3.5","created":1785188629,"description":"voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...","context_length":32000,"architecture":{"modality":"text+image->embeddings","input_modalities":["text","image"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.26e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004032,"max_completion_cost":0.0,"max_cost":0.004032},"sats_pricing":{"prompt":0.00019380698992595879,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.201823677630682,"max_completion_cost":0.0,"max_cost":6.201823677630682},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-multimodal-3.5-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-lite","name":"VoyageAI by MongoDB: voyage-4-lite","created":1785188627,"description":"voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1000000000000003e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0006720000000000001,"max_completion_cost":0.0,"max_cost":0.0006720000000000001},"sats_pricing":{"prompt":3.230116498765981e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.0336372796051139,"max_completion_cost":0.0,"max_cost":1.0336372796051139},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-lite-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4","name":"VoyageAI by MongoDB: voyage-4","created":1785188626,"description":"voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.3e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002016,"max_completion_cost":0.0,"max_cost":0.002016},"sats_pricing":{"prompt":9.690349496297939e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.100911838815341,"max_completion_cost":0.0,"max_cost":3.100911838815341},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-large","name":"VoyageAI by MongoDB: voyage-4-large","created":1785188624,"description":"voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.26e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004032,"max_completion_cost":0.0,"max_cost":0.004032},"sats_pricing":{"prompt":0.00019380698992595879,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.201823677630682,"max_completion_cost":0.0,"max_cost":6.201823677630682},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-large-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2","name":"Google: Gemini Embedding 2","created":1779290135,"description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":0.0,"request":0.0,"image":4.725e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00172032,"max_completion_cost":0.0,"max_cost":0.00172032},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0,"request":0.001,"image":0.0007267762122223455,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.646111435789091,"max_completion_cost":0.0,"max_cost":2.646111435789091},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2-preview","name":"Google: Gemini Embedding 2 Preview","created":1776436465,"description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.1e-07,"completion":0.0,"request":0.0,"image":4.725e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00172032,"max_completion_cost":0.0,"max_cost":0.00172032},"sats_pricing":{"prompt":0.000323011649876598,"completion":0.0,"request":0.001,"image":0.0007267762122223455,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.646111435789091,"max_completion_cost":0.0,"max_cost":2.646111435789091},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2-preview","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-4b","name":"Perplexity: Embed V1 4B","created":1773625372,"description":"pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.15e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001008,"max_completion_cost":0.0,"max_cost":0.001008},"sats_pricing":{"prompt":4.8451747481489696e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.5504559194076706,"max_completion_cost":0.0,"max_cost":1.5504559194076706},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-4B","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-0.6b","name":"Perplexity: Embed V1 0.6B","created":1773624868,"description":"pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.2e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00013440000000000001,"max_completion_cost":0.0,"max_cost":0.00013440000000000001},"sats_pricing":{"prompt":6.460232997531961e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.20672745592102276,"max_completion_cost":0.0,"max_cost":0.20672745592102276},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-0.6B","alias_ids":null,"forwarded_model_id":null},{"id":"gte-base","name":"Thenlper: GTE-Base","created":1763433820,"description":"The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-base-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"gte-large","name":"Thenlper: GTE-Large","created":1763433655,"description":"The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.376000000000001e-06,"max_completion_cost":0.0,"max_cost":5.376000000000001e-06},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00826909823684091,"max_completion_cost":0.0,"max_cost":0.00826909823684091},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-large-v2","name":"Intfloat: E5-Large-v2","created":1763433432,"description":"The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.376000000000001e-06,"max_completion_cost":0.0,"max_cost":5.376000000000001e-06},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00826909823684091,"max_completion_cost":0.0,"max_cost":0.00826909823684091},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-large-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-base-v2","name":"Intfloat: E5-Base-v2","created":1763433192,"description":"The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multilingual-e5-large","name":"Intfloat: Multilingual-E5-Large","created":1763433047,"description":"The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.376000000000001e-06,"max_completion_cost":0.0,"max_cost":5.376000000000001e-06},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00826909823684091,"max_completion_cost":0.0,"max_cost":0.00826909823684091},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/multilingual-e5-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"paraphrase-minilm-l6-v2","name":"Sentence Transformers: paraphrase-MiniLM-L6-v2","created":1763432454,"description":"The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/paraphrase-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l12-v2","name":"Sentence Transformers: all-MiniLM-L12-v2","created":1763432155,"description":"The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l12-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-base-en-v1.5","name":"BAAI: bge-base-en-v1.5","created":1763431837,"description":"The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-base-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multi-qa-mpnet-base-dot-v1","name":"Sentence Transformers: multi-qa-mpnet-base-dot-v1","created":1763431339,"description":"The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-large-en-v1.5","name":"BAAI: bge-large-en-v1.5","created":1763431087,"description":"The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.376000000000001e-06,"max_completion_cost":0.0,"max_cost":5.376000000000001e-06},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00826909823684091,"max_completion_cost":0.0,"max_cost":0.00826909823684091},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-large-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-m3","name":"BAAI: bge-m3","created":1763424372,"description":"The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.","context_length":8194,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.601600000000001e-05,"max_completion_cost":0.0,"max_cost":8.601600000000001e-05},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.13230557178945457,"max_completion_cost":0.0,"max_cost":0.13230557178945457},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-m3-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-mpnet-base-v2","name":"Sentence Transformers: all-mpnet-base-v2","created":1763421830,"description":"The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-mpnet-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l6-v2","name":"Sentence Transformers: all-MiniLM-L6-v2","created":1763421176,"description":"The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.250000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.6880000000000004e-06,"max_completion_cost":0.0,"max_cost":2.6880000000000004e-06},"sats_pricing":{"prompt":8.075291246914952e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004134549118420455,"max_completion_cost":0.0,"max_cost":0.004134549118420455},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-embed-2312","name":"Mistral: Mistral Embed 2312","created":1761944622,"description":"Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00086016,"max_completion_cost":0.0,"max_cost":0.00086016},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3230557178945455,"max_completion_cost":0.0,"max_cost":1.3230557178945455},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-embed-2312","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-001","name":"Google: Gemini Embedding 001","created":1761943410,"description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","context_length":20000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00315,"max_completion_cost":0.0,"max_cost":0.00315},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.84517474814897,"max_completion_cost":0.0,"max_cost":4.84517474814897},"per_request_limits":null,"top_provider":{"context_length":20000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-001","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-ada-002","name":"OpenAI: Text Embedding Ada 002","created":1761865798,"description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.05e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00086016,"max_completion_cost":0.0,"max_cost":0.00086016},"sats_pricing":{"prompt":0.000161505824938299,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3230557178945455,"max_completion_cost":0.0,"max_cost":1.3230557178945455},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-ada-002","alias_ids":["text-embedding-ada-002-v2"],"forwarded_model_id":null},{"id":"codestral-embed-2505","name":"Mistral: Codestral Embed 2505","created":1761864460,"description":"Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.575e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00129024,"max_completion_cost":0.0,"max_cost":0.00129024},"sats_pricing":{"prompt":0.00024225873740744852,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9845835768418183,"max_completion_cost":0.0,"max_cost":1.9845835768418183},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/codestral-embed-2505","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-large","name":"OpenAI: Text Embedding 3 Large","created":1761862866,"description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.365e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001118208,"max_completion_cost":0.0,"max_cost":0.001118208},"sats_pricing":{"prompt":0.00020995757241978874,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.7199724332629094,"max_completion_cost":0.0,"max_cost":1.7199724332629094},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-3-large","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-small","name":"OpenAI: Text Embedding 3 Small","created":1761857455,"description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1000000000000003e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00017203200000000002,"max_completion_cost":0.0,"max_cost":0.00017203200000000002},"sats_pricing":{"prompt":3.230116498765981e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.26461114357890914,"max_completion_cost":0.0,"max_cost":0.26461114357890914},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-3-small","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-8b","name":"Qwen: Qwen3 Embedding 8B","created":1761680622,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0500000000000001e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00033600000000000004,"max_completion_cost":0.0,"max_cost":0.00033600000000000004},"sats_pricing":{"prompt":1.6150582493829903e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5168186398025569,"max_completion_cost":0.0,"max_cost":0.5168186398025569},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-4b","name":"Qwen: Qwen3 Embedding 4B","created":1761662922,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1000000000000003e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0006881280000000001,"max_completion_cost":0.0,"max_cost":0.0006881280000000001},"sats_pricing":{"prompt":3.230116498765981e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.0584445743156365,"max_completion_cost":0.0,"max_cost":1.0584445743156365},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-4b","alias_ids":null,"forwarded_model_id":null}]}