{"object":"list","input_price_multiplier":0.32,"cached_input_price_multiplier":0.128,"cached_input_price_multiplier_credits":0.128,"cached_input_price_multiplier_subscription":0.32,"request_multiplier_gates":[{"minMultiplier":201,"plans":["enterprise"]},{"minMultiplier":101,"plans":["ultra","enterprise"]},{"minMultiplier":51,"plans":["pro","ultra","enterprise"]},{"minMultiplier":9,"plans":["plus","pro","ultra","enterprise"]},{"minMultiplier":5,"plans":["basic","plus","pro","ultra","enterprise"]}],"subscription_plans":[{"key":"micro","name":"Composite Micro","level":"micro-new","paypalPlanId":"P-6XX78334BJ348300CNKOW2XA","dailyRequests":100,"freeRpd":250,"maxMultiplier":4,"dailyBits":3000,"dailySpendCeiling":0.2,"periodSpendCeiling":3.75,"requestTokens":13000},{"key":"basic","name":"Composite Basic","level":"basic-new","paypalPlanId":"P-6CP33042HB3913132NI5CE4Y","dailyRequests":200,"freeRpd":500,"maxMultiplier":8,"dailyBits":5000,"dailySpendCeiling":0.33,"periodSpendCeiling":6.25,"requestTokens":11000},{"key":"basic-coding","name":"Composite Basic (Coding)","level":"basic-coding-new","paypalPlanId":"P-6B516438FM798321UNI5LIPQ","dailyRequests":200,"freeRpd":500,"maxMultiplier":8,"dailyBits":5000,"dailySpendCeiling":0.4,"periodSpendCeiling":7.5,"requestTokens":13000},{"key":"plus","name":"Composite Plus","level":"plus-new","paypalPlanId":"P-9TM745583H737263LNI5CFGY","dailyRequests":500,"freeRpd":null,"maxMultiplier":50,"dailyBits":10000,"dailySpendCeiling":0.65,"periodSpendCeiling":12.5,"requestTokens":8000},{"key":"plus-coding","name":"Composite Plus (Coding)","level":"plus-coding-new","paypalPlanId":"P-9JV37179P9682940PNI5LJAY","dailyRequests":500,"freeRpd":null,"maxMultiplier":50,"dailyBits":10000,"dailySpendCeiling":0.73,"periodSpendCeiling":13.75,"requestTokens":9000},{"key":"pro","name":"Composite Pro","level":"pro-new","paypalPlanId":"P-13X26467KY7913042NI5CFOI","dailyRequests":1000,"freeRpd":null,"maxMultiplier":100,"dailyBits":20000,"dailySpendCeiling":1,"periodSpendCeiling":18.75,"requestTokens":6000},{"key":"pro-coding","name":"Composite Pro (Coding)","level":"pro-coding-new","paypalPlanId":"P-9SP3732093694534NNI5LIYY","dailyRequests":1000,"freeRpd":null,"maxMultiplier":100,"dailyBits":20000,"dailySpendCeiling":1.07,"periodSpendCeiling":20,"requestTokens":7000},{"key":"ultra","name":"Composite Ultra","level":"ultra-new","paypalPlanId":"P-82F57140LV0369122NKOW2NI","dailyRequests":2000,"freeRpd":null,"maxMultiplier":200,"dailyBits":30000,"dailySpendCeiling":1.3,"periodSpendCeiling":25,"requestTokens":4000},{"key":"enterprise","name":"Composite Enterprise","level":"enterprise-new","paypalPlanId":"P-51F61065YU795733HNKHFH4A","dailyRequests":5000,"freeRpd":null,"maxMultiplier":null,"dailyBits":50000,"dailySpendCeiling":3.3,"periodSpendCeiling":62.5,"requestTokens":4000}],"data":[{"id":"composite/anthropic-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000005","completion":"0.000025"},"name":"Composite Anthropic Router","router":true,"ensemble":false,"request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"composite/gemini-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000001","completion":"0.000006"},"name":"Composite Gemini Router","router":true,"ensemble":false,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"composite/gpt-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000001","completion":"0.000006"},"name":"Composite GPT Router","router":true,"ensemble":false,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"composite/planning-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"3.75e-7","completion":"0.00000225"},"name":"Composite Planning Router","router":true,"ensemble":false,"request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"composite/execution-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"3.75e-7","completion":"0.00000225"},"name":"Composite Execution Router","router":true,"ensemble":false,"request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"composite/subagent-router","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"1.19e-7","completion":"4e-7"},"name":"Composite Subagent Router","router":true,"ensemble":false,"request_multiplier":1,"required_plans":null},{"id":"lucidityai/leviathan-2-flash","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.0000019025","completion":"0.00000883"},"name":"Leviathan 2 Flash","router":true,"ensemble":true,"request_multiplier":24,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"lucidityai/leviathan-2-base","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000005600000000000001","completion":"0.000024025"},"name":"Leviathan 2 Base","router":true,"ensemble":true,"request_multiplier":68,"required_plans":["pro","ultra","enterprise"]},{"id":"lucidityai/leviathan-2-pro","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000009075","completion":"0.00004335000000000001"},"name":"Leviathan 2 Pro","router":true,"ensemble":true,"request_multiplier":118,"required_plans":["ultra","enterprise"]},{"id":"lucidityai/leviathan-2-ultra","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000019075000000000003","completion":"0.00009335"},"name":"Leviathan 2 Ultra","router":true,"ensemble":true,"request_multiplier":251,"required_plans":["enterprise"]},{"id":"z-ai/glm-5.3-prime-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.0000029049999999999997","completion":"0.0000045496000000000005"},"name":"Z.ai: GLM 5.3 Prime (Cheap)","base_model":"z-ai/glm-5.3-prime","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flashx-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"pricing":{"prompt":"4.75e-7","completion":"6.462500000000001e-7"},"name":"Z.ai: GLM 5.3 FlashX (Cheap)","base_model":"z-ai/glm-5.3-flashx","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.8-flash-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.8 Flash (Cheap)","base_model":"google/gemini-3.8-flash","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":false},"request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.3-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"5.250000000000001e-7","completion":"6.824400000000001e-7"},"name":"Z.ai: GLM 5.3 (Cheap)","base_model":"z-ai/glm-5.3","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"google/gemini-3.7-flash-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.7 Flash (Cheap)","base_model":"google/gemini-3.7-flash","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-opus-5-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 5 (Cheap)","base_model":"anthropic/claude-opus-5","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.6-flash-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.6 Flash (Cheap)","base_model":"google/gemini-3.6-flash","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.2-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"6.675e-7","completion":"9.306e-7"},"name":"Z.ai: GLM 5.2 (Cheap)","base_model":"z-ai/glm-5.2","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":false},"request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.8 (Cheap)","base_model":"anthropic/claude-opus-4.8","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"8.550000000000001e-7","completion":"0.00000216"},"name":"Google: Gemini 3.5 Flash (Cheap)","base_model":"google/gemini-3.5-flash","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.7 (Cheap)","base_model":"anthropic/claude-opus-4.7","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000001071","completion":"0.0000015696119999999999"},"name":"Z.ai: GLM 5.1 (Cheap)","base_model":"z-ai/glm-5.1","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.6 (Cheap)","base_model":"anthropic/claude-opus-4.6","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":false},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.5 (Cheap)","base_model":"anthropic/claude-opus-4.5","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flash-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"pricing":{"prompt":"2.55e-7","completion":"2.585e-7"},"name":"Z.ai: GLM 5.3 Flash (normal) (Cheap)","base_model":"z-ai/glm-5.3-flash-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"1.7500000000000002e-7","completion":"0.000003619"},"name":"Z.ai: GLM 5.3 (normal) (Cheap)","base_model":"z-ai/glm-5.3-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 5 (normal) (Cheap)","base_model":"anthropic/claude-opus-5-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"2.5700000000000004e-7","completion":"0.000006204"},"name":"Z.ai: GLM 5.2 (normal) (Cheap)","base_model":"z-ai/glm-5.2-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.8 (normal) (Cheap)","base_model":"anthropic/claude-opus-4.8-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.7 (normal) (Cheap)","base_model":"anthropic/claude-opus-4.7-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000001505","completion":"0.0000022748000000000002"},"name":"Z.ai: GLM 5.1 (normal) (Cheap)","base_model":"z-ai/glm-5.1-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.6 (normal) (Cheap)","base_model":"anthropic/claude-opus-4.6-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-normal-cheap","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.5 (normal) (Cheap)","base_model":"anthropic/claude-opus-4.5-normal","modifier":"cheap","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"unbiased/pareto-26.10-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000032","input_cache_read":"0.00000003"},"name":"Pareto 26.10 Preview","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks. This is a preview of the...","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6.1-sol-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.00000005","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000001","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6.1 Sol Pro","description":"GPT-6.1 Sol Pro is the same underlying model as [GPT-6.1 Sol](https://openrouter.ai/openai/gpt-6.1-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6.1-sol","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.00000005","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000001","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6.1 Sol","description":"GPT-6.1 Sol is an upgrade to GPT-6 Sol from OpenAI, positioned below the flagship GPT-6 Astra in the GPT-6 series. It is suited for agentic coding, computer use, document-heavy professional...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5.5","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perceptron/perceptron-mk1.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":36864,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"name":"Perceptron: Perceptron Mk1.5","description":"Perceptron Mk1.5 is Perceptron's embodied reasoning model for physical agents. It accepts text, image, video, and audio input, and answers with text plus optional structured annotations: points, boxes, polygons, tracks,...","request_multiplier":3,"required_plans":null},{"id":"fireworks/ember-1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","input_cache_read":"0.0000003"},"name":"Fireworks: Ember-1","description":"Ember-1 is a specialized reasoning model from Fireworks Research, built on [Kimi K3](https://openrouter.ai/moonshotai/kimi-k3). It is designed to make every token go further: it produces shorter reasoning traces, using roughly 40%...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-prime","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000028","completion":"0.0000088","input_cache_read":"0.00000056"},"name":"Z.ai: GLM 5.3 Prime","description":"GLM-5.3-Prime is the high-speed variant of Z.ai's GLM-5.3, inheriting its full capabilities while delivering 1.5–2× the output throughput through inference acceleration. It supports text input and output with a 1M-token...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-max-prime","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000005"},"name":"Qwen: Qwen3.8 Max Prime","description":"Qwen3.8 Max Prime is a higher-throughput variant of Qwen3.8 Max from Alibaba's Qwen team, served as a separate SKU at a higher price point. It accepts text, image, and video...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.5-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"name":"AionLabs: Aion 3.5 Mini","description":"Aion 3.5 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It is the smaller, lower-cost sibling of Aion 3.5 and uses...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"name":"AionLabs: Aion 3.5","description":"Aion 3.5 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each...","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"upstage/solar-mini4","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.000000005"},"name":"Upstage: Solar Mini 4","description":"Solar Mini 4 is Upstage's compact, cost-efficient language model, a 35B-parameter mixture-of-experts with 3B active parameters and a 524K context window. It is built for agentic use cases where response...","request_multiplier":1,"required_plans":null},{"id":"cohere/command-a-plus","object":"model","owned_by":"cv11","created":1791290914,"context_length":192000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000015","input_cache_read":"0.00000015"},"name":"Cohere: Command A+","description":"Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured...","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-6-luna-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna Pro","description":"GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-luna","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","provider":"OpenAI","flex":true,"request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-sol-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol Pro","description":"GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6-sol","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"name":"Anthropic: Claude Opus 5.5","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000435","completion":"0.0000087","input_cache_read":"0.000000036"},"name":"Xiaomi: MiMo-V2.6-Pro-UltraSpeed","description":"MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x...","request_multiplier":36,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","request_multiplier":4,"required_plans":null},{"id":"x-ai/grok-4.7","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-omni-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016"},"name":"Qwen: Qwen3.8 Omni Flash","description":"Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,...","request_multiplier":2,"required_plans":null},{"id":"prism-ml/ternary-bonsai-2-27b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000005","input_cache_read":"0.0000000375"},"name":"PrismML: Ternary Bonsai 2 27B","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flashx","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000037","completion":"0.00000125","input_cache_read":"0.00000009"},"name":"Z.ai: GLM 5.3 FlashX","description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","request_multiplier":4,"required_plans":null},{"id":"unbiased/pareto","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.0000075","input_cache_read":"0.00000025"},"name":"Pareto","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks.","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inference-net/schematron-v2-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000015","input_cache_read":"0.00000003"},"name":"Inference.net: Schematron V2 Turbo","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","request_multiplier":1,"required_plans":null},{"id":"inference-net/schematron-v2-small","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000023","input_cache_read":"0.00000005"},"name":"Inference.net: Schematron V2 Small","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","request_multiplier":1,"required_plans":null},{"id":"sakana/fugu-ultra-v2","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"name":"Sakana: Fugu Ultra v2","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"sakana/fugu-max","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.01","input_cache_read":"0.00000025"},"name":"Sakana: Fugu Max","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash-vl","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000021","completion":"0.0000000616","input_cache_read":"0.0000000042"},"name":"inclusionAI: Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","request_multiplier":1,"required_plans":null},{"id":"inception/mercury-2.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":260000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000004","completion":"0.00000015","input_cache_read":"0.000000004"},"name":"Inception: Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000025","completion":"0.0000001","input_cache_read":"0.0000000025"},"name":"Nex AGI: Nex-N2.5-Mini","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"Nex AGI: Nex-N2.5-Pro","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-astra","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}],"discount":0},"name":"OpenAI: GPT-6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","provider":"OpenAI","flex":true,"request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"openai/gpt-6-astra-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}],"discount":0},"name":"OpenAI: GPT-6 Astra Pro","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-max-0902","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","input_cache_write":"0.0000025"},"name":"Qwen: Qwen3.8 Max (0902)","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta/muse-spark-1.3-contributor","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"name":"Meta: Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","request_multiplier":1,"required_plans":null},{"id":"meta/muse-spark-1.3","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.8-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0.5},"name":"Google: Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","provider":"Google AI Studio","flex":true,"request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"ibm-granite/granite-4.2-8b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"IBM: Granite 4.2 8B","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","request_multiplier":1,"required_plans":null},{"id":"tencent/hy4-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash-fin","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.0000001232","input_cache_read":"0.0000000084"},"name":"inclusionAI: Ling 3.0 Flash Fin","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.8-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016","input_cache_write":"0.0000002"},"name":"Qwen: Qwen3.8 Flash","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015","discount":0.5},"name":"Z.ai: GLM 5.3 Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","request_multiplier":1,"required_plans":null},{"id":"meta/muse-spark-1.2-contributor","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"name":"Meta: Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","request_multiplier":2,"required_plans":null},{"id":"tencent/hy-mt2-1.8b","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","stop","temperature"],"pricing":{"prompt":"0.000000044","completion":"0.000000177"},"name":"Tencent: Hy-MT2-1.8B","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","request_multiplier":1,"required_plans":null},{"id":"tencent/hy-mt2-30b-a3b","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"name":"Tencent: Hy-MT2-30B-A3B","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","request_multiplier":1,"required_plans":null},{"id":"tencent/hy-mt2-7b","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"name":"Tencent: Hy-MT2-7B","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000132","input_cache_read":"0.000000078","discount":0.7},"name":"Z.ai: GLM 5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.00000053125","discount":0.25},"name":"Qwen: Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","request_multiplier":4,"required_plans":null},{"id":"google/gemini-3.7-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0.5},"name":"Google: Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","provider":"Google","flex":true,"request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-2-1-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000025"},"name":"ByteDance Seed: Seed 2.1 Turbo","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"name":"Qwen: Qwen3.8 2.4T A95B","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-2.0-code","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.000001","completion":"0.000006"}]},"name":"ByteDance Seed: Seed-2.0-Code","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}],"discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-lightning","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000049","completion":"0.00000014","input_cache_read":"0.0000000245","discount":0.3},"name":"NVIDIA: Nemotron 3.5 Lightning","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","request_multiplier":1,"required_plans":null},{"id":"sakana/sakana-namazu","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.00000095","completion":"0.000004","web_search":"0.007","input_cache_read":"0.00000015"},"name":"Sakana: Sakana Namazu","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"upstage/solar-pro4","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000036","input_cache_read":"0.000000018"},"name":"Upstage: Solar Pro 4","description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","request_multiplier":1,"required_plans":null},{"id":"meta/muse-glimmer-30b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","request_multiplier":3,"required_plans":null},{"id":"meta/muse-spark-1.2","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, and PDF documents, returns text, and offers a 1M-token context window....","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-flash-0731","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","request_multiplier":1,"required_plans":null},{"id":"thinkingmachines/inkling-small","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.0000012","input_cache_read":"0.0000001"},"name":"Thinking Machines: Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.7-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000013","input_cache_read":"0.000000006","input_cache_write":"0.000000038","overrides":[{"min_prompt_tokens":32000,"prompt":"0.0000001","completion":"0.0000004","input_cache_read":"0.00000002","input_cache_write":"0.000000125"},{"min_prompt_tokens":256000,"prompt":"0.0000002","completion":"0.0000008","input_cache_read":"0.00000004","input_cache_write":"0.00000025"}]},"name":"Qwen: Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-opus-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000021","completion":"0.000000063","input_cache_read":"0.0000000042"},"name":"inclusionAI: Ling 3.0 Flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","request_multiplier":1,"required_plans":null},{"id":"poolside/laguna-s-2.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000009"},"name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.6-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0},"name":"Google: Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","provider":"Google","flex":true,"request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash-lite","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.00000015","input_audio_cache":"0.000000015","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.000000015","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","provider":"Google","flex":true,"request_multiplier":3,"required_plans":null},{"id":"meituan/longcat-2.0","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048756,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.000000006"},"name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","request_multiplier":4,"required_plans":null},{"id":"thinkingmachines/inkling","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.00000405","input_cache_read":"0.00000016"},"name":"Thinking Machines: Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta/muse-spark-1.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, and PDF documents and returns text, with a 1M-token context window....","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"kwaipilot/kat-coder-pro-v2.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000074","completion":"0.00000296","input_cache_read":"0.00000015"},"name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-luna-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna Pro","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-luna","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","provider":"OpenAI","flex":true,"request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-terra-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra Pro","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-terra","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","provider":"OpenAI","flex":true,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol Pro","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","provider":"OpenAI","flex":true,"request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.0-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.0","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","request_multiplier":2,"required_plans":null},{"id":"poolside/laguna-xs-2.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000006","completion":"0.00000012","input_cache_read":"0.00000003"},"name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-sonnet-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.1-flash-lite-image","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","temperature","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000015","image_output":"0.00003","web_search":"0.014"},"name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","request_multiplier":4,"required_plans":null},{"id":"sakana/fugu-ultra","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"name":"Sakana: Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-3.1-flash-image","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3-pro-image","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","image_output":"0.00012","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","request_multiplier":30,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005625","completion":"0.0000018","input_cache_read":"0.000000105","discount":0.25},"name":"Z.ai: GLM 5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-content-safety","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002"},"name":"NVIDIA: Nemotron 3.5 Content Safety","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-ultra-550b-a55b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.7-plus","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000032","completion":"0.00000128","input_cache_read":"0.000000064","input_cache_write":"0.0000004","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000096","completion":"0.00000384","input_cache_read":"0.000000192","input_cache_write":"0.0000012"}]},"name":"Qwen: Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m3","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.00000096","input_cache_read":"0.00000005","discount":0},"name":"MiniMax: MiniMax M3","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","request_multiplier":3,"required_plans":null},{"id":"stepfun/step-3.7-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000115","input_cache_read":"0.00000004"},"name":"StepFun: Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","request_multiplier":3,"required_plans":null},{"id":"anthropic/claude-opus-4.8","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3.7-max","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001475","completion":"0.000004425","input_cache_read":"0.000000295","input_cache_write":"0.00000184375"},"name":"Qwen: Qwen3.7 Max","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000075","completion":"0.0000045","image":"0.00000075","audio":"0.0000015","input_audio_cache":"0.00000015","web_search":"0.014","internal_reasoning":"0.0000045","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.5 Flash","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","provider":"Google","flex":true,"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perceptron/perceptron-mk1","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"name":"Perceptron: Perceptron Mk1","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.1-flash-lite","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","provider":"Google","flex":true,"request_multiplier":2,"required_plans":null},{"id":"openai/gpt-chat-latest","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT Chat Latest","description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"name":"Mistral: Mistral Medium 3.5","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-plus-20260420","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000018","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000375","completion":"0.00000225","input_cache_write":"0.00000046875"}]},"name":"Qwen: Qwen3.5 Plus 2026-04-20","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.000001125","input_cache_write":"0.000000234375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000075","completion":"0.000003","input_cache_write":"0.0000009375"}]},"name":"Qwen: Qwen3.6 Flash","description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000007","input_cache_read":"0.000000025","discount":0},"name":"Qwen: Qwen3.6 35B A3B","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.6-max-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001027","completion":"0.000006162","input_cache_write":"0.00000128375","overrides":[{"min_prompt_tokens":128000,"prompt":"0.00000158","completion":"0.00000948","input_cache_write":"0.000001975"}]},"name":"Qwen: Qwen3.6 Max Preview","description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000002","input_cache_read":"0.00000003","discount":0},"name":"Qwen: Qwen3.6 27B","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.5-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}],"discount":0},"name":"OpenAI: GPT-5.5 Pro","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","provider":"OpenAI","flex":true,"request_multiplier":225,"required_plans":["enterprise"]},{"id":"openai/gpt-5.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}],"discount":0},"name":"OpenAI: GPT-5.5","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","provider":"OpenAI","flex":true,"request_multiplier":38,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"name":"DeepSeek: DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.000000084","input_cache_read":"0.0000000084","discount":0.7},"name":"DeepSeek: DeepSeek V4 Flash 0423","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","request_multiplier":1,"required_plans":null},{"id":"tencent/hy3-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.0000006","input_cache_read":"0.00000006"},"name":"Tencent: Hy3 preview","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","request_multiplier":2,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.4-image-2","object":"model","owned_by":"cv11","created":1791290914,"context_length":272000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","top_logprobs","verbosity"],"pricing":{"prompt":"0.000008","completion":"0.000015","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"name":"OpenAI: GPT-5.4 Image 2","description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","request_multiplier":65,"required_plans":["pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000465","completion":"0.00000245","input_cache_read":"0.0000000975","discount":0},"name":"MoonshotAI: Kimi K2.6","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000966","completion":"0.000003036","input_cache_read":"0.0000001794","discount":0.31},"name":"Z.ai: GLM 5.1","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemma-4-26b-a4b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.00000022","input_cache_read":"0.000000021","discount":0},"name":"Google: Gemma 4 26B A4B ","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 31B","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:free","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0","completion":"0"},"name":"Google: Gemma 4 31B (free)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","upstream_free":true,"request_multiplier":0,"required_plans":null},{"id":"qwen/qwen3.6-plus","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000325","completion":"0.00000195","input_cache_write":"0.00000040625","overrides":[{"min_prompt_tokens":256000,"prompt":"0.0000013","completion":"0.0000039","input_cache_write":"0.000001625"}]},"name":"Qwen: Qwen3.6 Plus","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5v-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"name":"Z.ai: GLM 5V Turbo","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"arcee-ai/trinity-large-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000008","input_cache_read":"0.00000006"},"name":"Arcee AI: Trinity Large Thinking","description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","request_multiplier":3,"required_plans":null},{"id":"x-ai/grok-4.20-multi-agent","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"rekaai/reka-edge","object":"model","owned_by":"cv11","created":1791290914,"context_length":16384,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001"},"name":"Reka Edge","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2.7","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042"},"name":"MiniMax: MiniMax M2.7","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-nano","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.000000625","web_search":"0.01","input_cache_read":"0.00000001","discount":0},"name":"OpenAI: GPT-5.4 Nano","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","provider":"OpenAI","flex":true,"request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000375","completion":"0.00000225","web_search":"0.01","input_cache_read":"0.0000000375","discount":0},"name":"OpenAI: GPT-5.4 Mini","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","provider":"OpenAI","flex":true,"request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-small-2603","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Mistral: Mistral Small 4","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"name":"Z.ai: GLM 5 Turbo","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-super-120b-a12b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000045"},"name":"NVIDIA: Nemotron 3 Super","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-super-120b-a12b:free","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0","completion":"0"},"name":"NVIDIA: Nemotron 3 Super (free)","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","upstream_free":true,"request_multiplier":0,"required_plans":null},{"id":"bytedance-seed/seed-2.0-lite","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"name":"ByteDance Seed: Seed-2.0-Lite","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-9b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.4-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}],"discount":0},"name":"OpenAI: GPT-5.4 Pro","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","provider":"OpenAI","flex":true,"request_multiplier":225,"required_plans":["enterprise"]},{"id":"openai/gpt-5.4","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000025","completion":"0.00001125","input_cache_read":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.4","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","provider":"OpenAI","flex":true,"request_multiplier":19,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inception/mercury-2","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"name":"Inception: Mercury 2","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.1-flash-lite-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite Preview","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","provider":"Google AI Studio","flex":true,"request_multiplier":2,"required_plans":null},{"id":"bytedance-seed/seed-2.0-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000002","completion":"0.0000008"}]},"name":"ByteDance Seed: Seed-2.0-Mini","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.1-flash-image-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-35b-a3b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004"},"name":"Qwen: Qwen3.5-35B-A3B","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-27b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"name":"Qwen: Qwen3.5-27B","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-122b-a10b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"name":"Qwen: Qwen3.5-122B-A10B","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-flash-02-23","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000065","completion":"0.00000026"},"name":"Qwen: Qwen3.5-Flash","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.1-pro-preview-customtools","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","audio","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018","audio":"0.000004","input_audio_cache":"0.0000004","input_cache_read":"0.0000004"}]},"name":"Google: Gemini 3.1 Pro Preview Custom Tools","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","request_multiplier":30,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-2.0","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000016","input_cache_read":"0.0000002"},"name":"AionLabs: Aion-2.0","description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.1-pro-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000006","image":"0.000001","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.000006","input_cache_read":"0.0000001","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000009","audio":"0.000002","input_audio_cache":"0.0000002","input_cache_read":"0.0000002"}],"discount":0},"name":"Google: Gemini 3.1 Pro Preview","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","provider":"Google","flex":true,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-plus-02-15","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000156","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000325","completion":"0.00000195"}]},"name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-397b-a17b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000039","completion":"0.00000234","input_cache_read":"0.00000022","discount":0},"name":"Qwen: Qwen3.5 397B A17B","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000095","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","request_multiplier":3,"required_plans":null},{"id":"z-ai/glm-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"name":"Z.ai: GLM 5","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-max-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000078","completion":"0.0000039","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975"}]},"name":"Qwen: Qwen3 Max Thinking","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-next","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Coder Next","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","request_multiplier":2,"required_plans":null},{"id":"stepfun/step-3.5-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"StepFun: Step 3.5 Flash","description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007"},"name":"MoonshotAI: Kimi K2.5","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"upstage/solar-pro-3","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Upstage: Solar Pro 3","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2-her","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2-her","description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","request_multiplier":4,"required_plans":null},{"id":"writer/palmyra-x5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1040000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.000006"},"name":"Writer: Palmyra X5","description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-audio","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","audio"],"output_modalities":["text","audio"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001","audio":"0.000032","audio_output":"0.000064"},"name":"OpenAI: GPT Audio","description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-audio-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","audio"],"output_modalities":["text","audio"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000024","audio":"0.0000006","audio_output":"0.0000024"},"name":"OpenAI: GPT Audio Mini","description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.2-codex","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.2-Codex","description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-1.6-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000001","completion":"0.0000008"}]},"name":"ByteDance Seed: Seed 1.6 Flash","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","request_multiplier":1,"required_plans":null},{"id":"bytedance-seed/seed-1.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"name":"ByteDance Seed: Seed 1.6","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2.1","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-4.7","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.7","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3-flash-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3 Flash Preview","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","provider":"Google","flex":true,"request_multiplier":4,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.2-chat","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.2 Chat","description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.2-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000021","completion":"0.000168","web_search":"0.01"},"name":"OpenAI: GPT-5.2 Pro","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","request_multiplier":385,"required_plans":["enterprise"]},{"id":"openai/gpt-5.2","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000875","completion":"0.000007","web_search":"0.01","input_cache_read":"0.0000000875","discount":0},"name":"OpenAI: GPT-5.2","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","provider":"OpenAI","flex":true,"request_multiplier":16,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/devstral-2512","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Devstral 2 2512","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"relace/relace-search","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000003"},"name":"Relace: Relace Search","description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6v","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000005"},"name":"Z.ai: GLM 4.6V","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","request_multiplier":3,"required_plans":null},{"id":"openai/gpt-5.1-codex-max","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5.1-Codex-Max","description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"amazon/nova-2-lite-v1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000025"},"name":"Amazon: Nova 2 Lite","description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/ministral-14b-2512","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"name":"Mistral: Ministral 3 14B 2512","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"name":"Mistral: Ministral 3 8B 2512","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"name":"Mistral: Ministral 3 3B 2512","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-large-2512","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000015","input_cache_read":"0.00000005"},"name":"Mistral: Mistral Large 3 2512","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000003096","input_cache_read":"0.0000000216","discount":0.28},"name":"DeepSeek: DeepSeek V3.2","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-4.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.5","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-3-pro-image-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000006","image":"0.000001","image_output":"0.00006","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.000006","input_cache_read":"0.0000001","input_cache_write":"0.0000001875","discount":0},"name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","provider":"Google AI Studio","flex":true,"request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000625","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000000625","discount":0},"name":"OpenAI: GPT-5.1","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","provider":"OpenAI","flex":true,"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-codex","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.00000013"},"name":"OpenAI: GPT-5.1-Codex","description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-codex-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.00000003"},"name":"OpenAI: GPT-5.1-Codex-Mini","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"name":"MoonshotAI: Kimi K2 Thinking","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"amazon/nova-premier-v1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.0000125","input_cache_read":"0.000000625"},"name":"Amazon: Nova Premier 1.0","description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","request_multiplier":33,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-pro-search","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.018"},"name":"Perplexity: Sonar Pro Search","description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/voxtral-small-24b-2507","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"name":"Mistral: Voxtral Small 24B 2507","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-safeguard-20b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000003","input_cache_read":"0.0000000375"},"name":"OpenAI: gpt-oss-safeguard-20b","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-vl-32b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000104","completion":"0.000000416"},"name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","request_multiplier":1,"required_plans":null},{"id":"ibm-granite/granite-4.0-h-micro","object":"model","owned_by":"cv11","created":1791290914,"context_length":131000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000017","completion":"0.000000112"},"name":"IBM: Granite 4.0 Micro","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5-image-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"pricing":{"prompt":"0.0000025","completion":"0.000002","image_output":"0.000008","web_search":"0.01","input_cache_read":"0.00000025"},"name":"OpenAI: GPT-5 Image Mini","description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","request_multiplier":16,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"name":"Anthropic: Claude Haiku 4.5","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-8b-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.0000021"},"name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-vl-8b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5-image","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00001","image_output":"0.00004","web_search":"0.01","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-5 Image","description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-2.5-flash-image","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","image_output":"0.00003","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"name":"Google: Nano Banana (Gemini 2.5 Flash Image)","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00012","web_search":"0.01"},"name":"OpenAI: GPT-5 Pro","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","request_multiplier":275,"required_plans":["enterprise"]},{"id":"z-ai/glm-4.6","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008"},"name":"Z.ai: GLM 4.6","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2-exp","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000041"},"name":"DeepSeek: DeepSeek V3.2 Exp","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","request_multiplier":2,"required_plans":null},{"id":"thedrummer/cydonia-24b-v4.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000005","input_cache_read":"0.00000015"},"name":"TheDrummer: Cydonia 24B V4.1","description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","request_multiplier":2,"required_plans":null},{"id":"relace/relace-apply-3","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","seed","stop"],"pricing":{"prompt":"0.00000085","completion":"0.00000125"},"name":"Relace: Relace Apply 3","description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000004"},"name":"Qwen: Qwen3 VL 235B A22B Thinking","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000088","input_cache_read":"0.00000011","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-max","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000078","completion":"0.0000039","input_cache_read":"0.000000156","input_cache_write":"0.000000975","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078","input_cache_read":"0.000000312","input_cache_write":"0.00000195"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"name":"Qwen: Qwen3 Max","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-plus","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000325","input_cache_read":"0.00000013","input_cache_write":"0.0000008125","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000117","completion":"0.00000585","input_cache_read":"0.000000234","input_cache_write":"0.0000014625"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"name":"Qwen: Qwen3 Coder Plus","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.1-terminus","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.000001"},"name":"DeepSeek: DeepSeek V3.1 Terminus","description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.000000975","input_cache_read":"0.000000039","input_cache_write":"0.00000024375","overrides":[{"min_prompt_tokens":32000,"prompt":"0.000000325","completion":"0.000001625","input_cache_read":"0.000000065","input_cache_write":"0.00000040625"},{"min_prompt_tokens":128000,"prompt":"0.00000052","completion":"0.0000026","input_cache_read":"0.000000104","input_cache_write":"0.00000065"}]},"name":"Qwen: Qwen3 Coder Flash","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-thinking","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000012"},"name":"Qwen: Qwen3 Next 80B A3B Thinking","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000975","completion":"0.00000078","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen-plus-2025-07-28","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000078","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234"}]},"name":"Qwen: Qwen Plus 0728","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","request_multiplier":3,"required_plans":null},{"id":"moonshotai/kimi-k2-0905","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"name":"MoonshotAI: Kimi K2 0905","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-30b-a3b-thinking-2507","object":"model","owned_by":"cv11","created":1791290914,"context_length":81920,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-4-405b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000003"},"name":"Nous: Hermes 4 405B","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-chat-v3.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","request_multiplier":3,"required_plans":null},{"id":"mistralai/mistral-medium-3.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3.1","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.5v","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000018","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.5V","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000125","completion":"0.000001","web_search":"0.01","input_cache_read":"0.0000000125","discount":0},"name":"OpenAI: GPT-5 Mini","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","provider":"OpenAI","flex":true,"request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5-nano","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000025","completion":"0.0000002","web_search":"0.01","input_cache_read":"0.0000000025","discount":0},"name":"OpenAI: GPT-5 Nano","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","provider":"OpenAI","flex":true,"request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000017","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: gpt-oss-120b","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"name":"OpenAI: gpt-oss-20b","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-opus-4.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000015","completion":"0.000075","web_search":"0.01","input_cache_read":"0.0000015","input_cache_write":"0.00001875","input_cache_write_1h":"0.00003"},"name":"Anthropic: Claude Opus 4.1","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","request_multiplier":200,"required_plans":["ultra","enterprise"]},{"id":"mistralai/codestral-2508","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000003"},"name":"Mistral: Codestral 2508","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000027","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004815","completion":"0.00000019305","discount":0.55},"name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.5","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.5","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.5-air","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"name":"Z.ai: GLM 4.5 Air","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-thinking-2507","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023"},"name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","request_multiplier":3,"required_plans":null},{"id":"bytedance/ui-tars-1.5-7b","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","input_cache_read":"0.0000001"},"name":"ByteDance: UI-TARS 7B ","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","request_multiplier":1,"required_plans":null},{"id":"google/gemini-2.5-flash-lite","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","image":"0.00000005","audio":"0.00000015","input_audio_cache":"0.000000015","web_search":"0.014","internal_reasoning":"0.0000002","input_cache_read":"0.000000005","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 2.5 Flash Lite","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","provider":"Google AI Studio","flex":true,"request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000057","completion":"0.0000023"},"name":"MoonshotAI: Kimi K2 0711","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000009"},"name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","request_multiplier":3,"required_plans":null},{"id":"tencent/hunyuan-a13b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000057"},"name":"Tencent: Hunyuan A13B Instruct","description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","request_multiplier":2,"required_plans":null},{"id":"morph/morph-v3-large","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["logprobs","max_tokens","response_format","stop","structured_outputs","temperature","top_logprobs"],"pricing":{"prompt":"0.0000009","completion":"0.0000019"},"name":"Morph: Morph V3 Large","description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"morph/morph-v3-fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":81920,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","stop","temperature"],"pricing":{"prompt":"0.0000008","completion":"0.0000012"},"name":"Morph: Morph V3 Fast","description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"baidu/ernie-4.5-vl-424b-a47b","object":"model","owned_by":"cv11","created":1791290914,"context_length":123000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000125"},"name":"Baidu: ERNIE 4.5 VL 424B A47B ","description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","request_multiplier":4,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000002","discount":0},"name":"Mistral: Mistral Small 3.2 24B","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000022"},"name":"MiniMax: MiniMax M1","description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-2.5-flash","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["file","image","text","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.000000015","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 2.5 Flash","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","provider":"Google AI Studio","flex":true,"request_multiplier":3,"required_plans":null},{"id":"google/gemini-2.5-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000005","image":"0.000000625","audio":"0.000000625","input_audio_cache":"0.0000000625","web_search":"0.014","internal_reasoning":"0.000005","input_cache_read":"0.0000000625","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.00000125","completion":"0.0000075","audio":"0.00000125","input_audio_cache":"0.000000125","input_cache_read":"0.000000125"}],"discount":0},"name":"Google: Gemini 2.5 Pro","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","provider":"Google AI Studio","flex":true,"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o3-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","file","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.00002","completion":"0.00008","web_search":"0.01"},"name":"OpenAI: o3 Pro","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","request_multiplier":233,"required_plans":["enterprise"]},{"id":"google/gemini-2.5-pro-preview","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["file","image","text","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000005","image":"0.000000625","audio":"0.000000625","input_audio_cache":"0.0000000625","web_search":"0.014","internal_reasoning":"0.000005","input_cache_read":"0.0000000625","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.00000125","completion":"0.0000075","audio":"0.00000125","input_audio_cache":"0.000000125","input_cache_read":"0.000000125"}],"discount":0},"name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","provider":"Google AI Studio","flex":true,"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"name":"DeepSeek: R1 0528","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"meta-llama/llama-guard-4-12b","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"name":"Meta: Llama Guard 4 12B","description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000005"},"name":"Qwen: Qwen3 30B A3B","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-8b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"name":"Qwen: Qwen3 8B","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"name":"Qwen: Qwen3 32B","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000455","completion":"0.00000182"},"name":"Qwen: Qwen3 235B A22B","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/o4-mini-high","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"name":"OpenAI: o4 Mini High","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o3","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: o3","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o4-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"name":"OpenAI: o4 Mini","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT-4.1","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"name":"OpenAI: GPT-4.1 Mini","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-nano","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"name":"OpenAI: GPT-4.1 Nano","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-maverick","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525"},"name":"Meta: Llama 4 Maverick","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout","object":"model","owned_by":"cv11","created":1791290914,"context_length":1310720,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Meta: Llama 4 Scout","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-chat-v3-0324","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000029","completion":"0.00000114","input_cache_read":"0.00000011"},"name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","request_multiplier":3,"required_plans":null},{"id":"openai/o1-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"pricing":{"prompt":"0.00015","completion":"0.0006","web_search":"0.01"},"name":"OpenAI: o1-pro","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","request_multiplier":1750,"required_plans":["enterprise"]},{"id":"mistralai/mistral-small-3.1-24b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000351","completion":"0.000000555"},"name":"Mistral: Mistral Small 3.1 24B","description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","request_multiplier":3,"required_plans":null},{"id":"google/gemma-3-4b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000001"},"name":"Google: Gemma 3 4B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-12b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000015"},"name":"Google: Gemma 3 12B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","request_multiplier":1,"required_plans":null},{"id":"cohere/command-a","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"name":"Cohere: Command A","description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"rekaai/reka-flash-3","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"Reka Flash 3","description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000016","input_cache_read":"0.00000004","discount":0},"name":"Google: Gemma 3 27B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","request_multiplier":1,"required_plans":null},{"id":"thedrummer/skyfall-36b-v2","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000008","input_cache_read":"0.00000025"},"name":"TheDrummer: Skyfall 36B V2","description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","request_multiplier":4,"required_plans":null},{"id":"perplexity/sonar-reasoning-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005"},"name":"Perplexity: Sonar Reasoning Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-pro","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005"},"name":"Perplexity: Sonar Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-deep-research","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005","internal_reasoning":"0.000003"},"name":"Perplexity: Sonar Deep Research","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-saba","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"name":"Mistral: Saba","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","request_multiplier":2,"required_plans":null},{"id":"openai/o3-mini-high","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"name":"OpenAI: o3 Mini High","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-rp-llama-3.1-8b","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000016"},"name":"AionLabs: Aion-RP 1.0 (8B)","description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen2.5-vl-72b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.000001","input_cache_read":"0.0000004"},"name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-plus","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_read":"0.000000052","input_cache_write":"0.000000325","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234","input_cache_read":"0.000000156","input_cache_write":"0.000000975"}]},"name":"Qwen: Qwen-Plus","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","request_multiplier":3,"required_plans":null},{"id":"openai/o3-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"name":"OpenAI: o3 Mini","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-small-24b-instruct-2501","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000008"},"name":"Mistral: Mistral Small 3","description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","request_multiplier":1,"required_plans":null},{"id":"perplexity/sonar","object":"model","owned_by":"cv11","created":1791290914,"context_length":127072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000001","completion":"0.000001","web_search":"0.005"},"name":"Perplexity: Sonar","description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1","object":"model","owned_by":"cv11","created":1791290914,"context_length":64000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000025"},"name":"DeepSeek: R1","description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-01","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000192,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000011"},"name":"MiniMax: MiniMax-01","description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","request_multiplier":3,"required_plans":null},{"id":"microsoft/phi-4","object":"model","owned_by":"cv11","created":1791290914,"context_length":16384,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000014"},"name":"Microsoft: Phi 4","description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-chat","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000002574","completion":"0.0000010287"},"name":"DeepSeek: DeepSeek V3","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","request_multiplier":3,"required_plans":null},{"id":"sao10k/l3.3-euryale-70b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000075"},"name":"Sao10K: Llama 3.3 Euryale 70B","description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/o1","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000015","completion":"0.00006","web_search":"0.01","input_cache_read":"0.0000075"},"name":"OpenAI: o1","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","request_multiplier":175,"required_plans":["ultra","enterprise"]},{"id":"cohere/command-r7b-12-2024","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000000375","completion":"0.00000015"},"name":"Cohere: Command R7B (12-2024)","description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000032","input_cache_read":"0.00000011","discount":0},"name":"Meta: Llama 3.3 70B Instruct","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-lite-v1","object":"model","owned_by":"cv11","created":1791290914,"context_length":300000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000024"},"name":"Amazon: Nova Lite 1.0","description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-micro-v1","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.000000035","completion":"0.00000014"},"name":"Amazon: Nova Micro 1.0","description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-pro-v1","object":"model","owned_by":"cv11","created":1791290914,"context_length":300000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000032"},"name":"Amazon: Nova Pro 1.0","description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4o-2024-11-20","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o (2024-11-20)","description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large 2407","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-2.5-coder-32b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000066","completion":"0.000001"},"name":"Qwen2.5 Coder 32B Instruct","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"thedrummer/unslopnemo-12b","object":"model","owned_by":"cv11","created":1791290914,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"name":"TheDrummer: UnslopNemo 12B","description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","request_multiplier":3,"required_plans":null},{"id":"anthracite-org/magnum-v4-72b","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.000005"},"name":"Magnum v4 72B","description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","request_multiplier":21,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-2.5-7b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.2-1b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":60000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.000000027","completion":"0.000000201"},"name":"Meta: Llama 3.2 1B Instruct","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.2-3b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000033"},"name":"Meta: Llama 3.2 3B Instruct","description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen-2.5-72b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000036","completion":"0.0000004"},"name":"Qwen2.5 72B Instruct","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","request_multiplier":2,"required_plans":null},{"id":"cohere/command-r-08-2024","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"name":"Cohere: Command R (08-2024)","description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","request_multiplier":2,"required_plans":null},{"id":"cohere/command-r-plus-08-2024","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"name":"Cohere: Command R+ (08-2024)","description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"sao10k/l3.1-euryale-70b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000085","completion":"0.00000085"},"name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-3-llama-3.1-70b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000007"},"name":"Nous: Hermes 3 70B Instruct","description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-3-llama-3.1-405b","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000001"},"name":"Nous: Hermes 3 405B Instruct","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"sao10k/l3-lunaris-8b","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"name":"Sao10K: Llama 3 8B Lunaris","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-2024-08-06","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o (2024-08-06)","description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta-llama/llama-3.1-70b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"name":"Meta: Llama 3.1 70B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000002","completion":"0.00000004","input_cache_read":"0.000000025","discount":0},"name":"Meta: Llama 3.1 8B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000003","discount":0},"name":"Mistral: Mistral Nemo","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-mini","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini-2024-07-18","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini (2024-07-18)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","request_multiplier":2,"required_plans":null},{"id":"google/gemma-2-27b-it","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000065"},"name":"Google: Gemma 2 27B","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-4o","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4o-2024-05-13","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.000015"},"name":"OpenAI: GPT-4o (2024-05-13)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","request_multiplier":50,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"microsoft/wizardlm-2-8x22b","object":"model","owned_by":"cv11","created":1791290914,"context_length":65535,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000062","completion":"0.00000062"},"name":"WizardLM-2 8x22B","description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-4-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00001","completion":"0.00003"},"name":"OpenAI: GPT-4 Turbo","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","request_multiplier":100,"required_plans":["pro","ultra","enterprise"]},{"id":"mistralai/mistral-large","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-0613","object":"model","owned_by":"cv11","created":1791290914,"context_length":4095,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002"},"name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-instruct","object":"model","owned_by":"cv11","created":1791290914,"context_length":4095,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.000002"},"name":"OpenAI: GPT-3.5 Turbo Instruct","description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-16k","object":"model","owned_by":"cv11","created":1791290914,"context_length":16385,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000004"},"name":"OpenAI: GPT-3.5 Turbo 16k","description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mancer/weaver","object":"model","owned_by":"cv11","created":1791290914,"context_length":8000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000075"},"name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","request_multiplier":3,"required_plans":null},{"id":"undi95/remm-slerp-l2-13b","object":"model","owned_by":"cv11","created":1791290914,"context_length":6144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000035","completion":"0.00000065"},"name":"ReMM SLERP 13B","description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"name":"MythoMax 13B","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-3.5-turbo","object":"model","owned_by":"cv11","created":1791290914,"context_length":16385,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000015"},"name":"OpenAI: GPT-3.5 Turbo","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4","object":"model","owned_by":"cv11","created":1791290914,"context_length":8191,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00003","completion":"0.00006"},"name":"OpenAI: GPT-4","description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","request_multiplier":250,"required_plans":["enterprise"]},{"id":"anthropic/claude-sonnet-5.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5.5 (fast)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5","provider":"Amazon Bedrock","variant":"fast","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5.5 (cheapest-provider)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5","provider":"Google","variant":"cheapest-provider","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5.5 (normal)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5","variant":"normal","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008","discount":0},"name":"Anthropic: Claude Opus 5.5 (fast)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5","provider":"Azure","variant":"fast","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008","discount":0},"name":"Anthropic: Claude Opus 5.5 (cheapest-provider)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5","provider":"Amazon Bedrock","variant":"cheapest-provider","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"name":"Anthropic: Claude Opus 5.5 (normal)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5","variant":"normal","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-flash:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (fast)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash","provider":"DeepInfra","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (cheapest-provider)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.000000003","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (quality)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash","provider":"GMICloud","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"name":"Xiaomi: MiMo-V2.6-Flash (normal)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (fast)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro","provider":"DeepInfra","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (cheapest-provider)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.000000004","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (quality)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro","provider":"GMICloud","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"name":"Xiaomi: MiMo-V2.6-Pro (normal)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"x-ai/grok-4.7:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (fast)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7","provider":"xAI","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.7:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (cheapest-provider)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7","provider":"xAI","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.7-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (normal)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4.1-flash:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (fast)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","provider":"Decart","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (cheapest-provider)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","provider":"Decart","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000098","completion":"0.00000039","input_cache_read":"0.000000009","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (quality)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","provider":"Morph","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000012563","completion":"0.00000132","input_cache_read":"0.000000012563"},"name":"DeepSeek: DeepSeek V4.1 Flash (normal)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-fable-5.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5.1 (fast)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1","provider":"Google","variant":"fast","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5.1 (cheapest-provider)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1","provider":"Azure","variant":"cheapest-provider","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5.1 (normal)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1","variant":"normal","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"tencent/hy4-preview:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (fast)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","provider":"SiliconFlow","variant":"fast","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (cheapest-provider)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (quality)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","provider":"SiliconFlow","variant":"quality","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview (normal)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","variant":"normal","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flash:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003","discount":0},"name":"Z.ai: GLM 5.3 Flash (fast)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash","provider":"BaseTen","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-flash:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015","discount":0.5},"name":"Z.ai: GLM 5.3 Flash (cheapest-provider)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flash:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000105","completion":"0.00000035","input_cache_read":"0.0000000245","discount":0.3},"name":"Z.ai: GLM 5.3 Flash (quality)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash","provider":"Near AI","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flash-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003"},"name":"Z.ai: GLM 5.3 Flash (normal)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048575,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000014","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (fast)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","provider":"GMICloud","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686","discount":0.51},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (cheapest-provider)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000028","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (quality)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","provider":"SiliconFlow","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (normal)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000084","completion":"0.00000264","input_cache_read":"0.000000156","discount":0.4},"name":"Z.ai: GLM 5.3 (fast)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3","provider":"Phala","variant":"fast","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000132","input_cache_read":"0.000000078","discount":0.7},"name":"Z.ai: GLM 5.3 (cheapest-provider)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3","provider":"Novita","variant":"cheapest-provider","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.3:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.3 (quality)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3","provider":"Parasail","variant":"quality","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.000007","input_cache_read":"0.000000065"},"name":"Z.ai: GLM 5.3 (normal)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3","variant":"normal","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-27b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (fast)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","provider":"DeepInfra","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (cheapest-provider)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","provider":"Phala","variant":"cheapest-provider","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (quality)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","provider":"DeepInfra","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000425","completion":"0.00000255","input_cache_read":"0.000000085","input_cache_write":"0.00000053125"},"name":"Qwen: Qwen3.8 27B (normal)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","variant":"normal","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (fast)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b","provider":"Novita","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (cheapest-provider)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b","provider":"Novita","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (quality)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b","provider":"SiliconFlow","variant":"quality","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"name":"Qwen: Qwen3.8 2.4T A95B (normal)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000957","completion":"0.0000028776","input_cache_read":"0.000000099","discount":0.34},"name":"DeepSeek: DeepSeek V4 Pro 0813 (fast)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","provider":"Phala","variant":"fast","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813 (cheapest-provider)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","provider":"Ionstream","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001056","completion":"0.000003168","input_cache_read":"0.000000035","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0813 (quality)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","provider":"NextBit","variant":"quality","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}]},"name":"DeepSeek: DeepSeek V4 Pro 0813 (normal)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","variant":"normal","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000022","completion":"0.0000066","web_search":"0.01","input_cache_read":"0.00000055","input_cache_write":"0","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000044","completion":"0.0000132","input_cache_read":"0.0000011","input_cache_write":"0"}]},"name":"SpaceXAI: Grok 4.6 (fast)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6","provider":"Amazon Bedrock","variant":"fast","request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6 (cheapest-provider)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6","provider":"xAI","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6 (normal)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-lightning:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000016","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3.5 Lightning (fast)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning","provider":"DeepInfra","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000049","completion":"0.00000014","input_cache_read":"0.0000000245","discount":0.3},"name":"NVIDIA: Nemotron 3.5 Lightning (cheapest-provider)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning","provider":"Io Net","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.0000002","input_cache_read":"0.00000004","discount":0},"name":"NVIDIA: Nemotron 3.5 Lightning (quality)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning","provider":"CoreWeave","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000016","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3.5 Lightning (normal)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"meta/muse-glimmer-30b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (fast)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b","provider":"Phala","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"meta/muse-glimmer-30b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (cheapest-provider)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b","provider":"Phala","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"meta/muse-glimmer-30b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (quality)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b","provider":"DeepInfra","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"meta/muse-glimmer-30b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000004"},"name":"Meta: Muse Glimmer 30B (normal)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.0000000238","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (fast)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","provider":"DigitalOcean","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731 (cheapest-provider)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (quality)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","provider":"Parasail","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000118","completion":"0.00000128","input_cache_read":"0.0000000118"},"name":"DeepSeek: DeepSeek V4 Flash 0731 (normal)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Opus 5 (fast)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5","provider":"Anthropic","variant":"fast","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-opus-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 5 (cheapest-provider)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5 (normal)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5","variant":"normal","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000201","completion":"0.00001005","input_cache_read":"0.000000201","discount":0.33},"name":"MoonshotAI: Kimi K3 (fast)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","provider":"Decart","variant":"fast","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3 (cheapest-provider)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","provider":"Phala","variant":"cheapest-provider","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001274","completion":"0.000013296","input_cache_read":"0.000000278","discount":0},"name":"MoonshotAI: Kimi K3 (quality)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","provider":"Morph","variant":"quality","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.000014","input_cache_read":"0.00000031"},"name":"MoonshotAI: Kimi K3 (normal)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","variant":"normal","request_multiplier":28,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (fast)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5","provider":"xAI","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (cheapest-provider)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5","provider":"xAI","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (normal)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000053","input_cache_read":"0.000000033","discount":0},"name":"Tencent: Hy3 (fast)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","provider":"DeepInfra","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","discount":0,"overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (cheapest-provider)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","provider":"Tencent","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000058","input_cache_read":"0.000000035","discount":0},"name":"Tencent: Hy3 (quality)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","provider":"GMICloud","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (normal)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-sonnet-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (fast)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5","provider":"Amazon Bedrock","variant":"fast","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (cheapest-provider)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5 (normal)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5","variant":"normal","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000118","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.2 (fast)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2","provider":"Cloudflare","variant":"fast","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005625","completion":"0.0000018","input_cache_read":"0.000000105","discount":0.25},"name":"Z.ai: GLM 5.2 (cheapest-provider)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000232","completion":"0.000003553","input_cache_read":"0.000000137","discount":0},"name":"Z.ai: GLM 5.2 (quality)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2","provider":"Morph","variant":"quality","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000152","completion":"0.000012","input_cache_read":"0.00000015"},"name":"Z.ai: GLM 5.2 (normal)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2","variant":"normal","request_multiplier":21,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code (fast)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code","provider":"StreamLake","variant":"fast","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code (cheapest-provider)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000085916","completion":"0.0000038","input_cache_read":"0.00000017993","discount":0},"name":"MoonshotAI: Kimi K2.7 Code (quality)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code","provider":"SiliconFlow","variant":"quality","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006712","completion":"0.00000335","input_cache_read":"0.00000018"},"name":"MoonshotAI: Kimi K2.7 Code (normal)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code","variant":"normal","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5 (fast)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5","provider":"Anthropic","variant":"fast","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5 (cheapest-provider)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5 (normal)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5","variant":"normal","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (fast)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","provider":"DeepInfra","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (cheapest-provider)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000003125","input_cache_read":"0.0000001875","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (quality)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","provider":"Venice","variant":"quality","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra (normal)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","variant":"normal","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000028","completion":"0.0000011","input_cache_read":"0.000000056","discount":0},"name":"MiniMax: MiniMax M3 (fast)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3","provider":"DeepInfra","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.00000096","input_cache_read":"0.00000005","discount":0},"name":"MiniMax: MiniMax M3 (cheapest-provider)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3","provider":"CoreWeave","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m3:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M3 (quality)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3","provider":"Parasail","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006"},"name":"MiniMax: MiniMax M3 (normal)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-opus-4.8:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (fast)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8","provider":"Claude Platform on AWS","variant":"fast","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (cheapest-provider)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8 (normal)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8","variant":"normal","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (fast)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1","provider":"xAI","variant":"fast","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (cheapest-provider)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1","provider":"xAI","variant":"cheapest-provider","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (normal)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1","variant":"normal","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (fast)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3","provider":"xAI","variant":"fast","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (cheapest-provider)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3","provider":"xAI","variant":"cheapest-provider","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (normal)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3","variant":"normal","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075","discount":0},"name":"Mistral: Mistral Medium 3.5 (fast)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5","provider":"Mistral","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075","discount":0},"name":"Mistral: Mistral Medium 3.5 (cheapest-provider)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5","provider":"Mistral","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"name":"Mistral: Mistral Medium 3.5 (normal)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-35b-a3b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.000001","discount":0},"name":"Qwen: Qwen3.6 35B A3B (fast)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b","provider":"Venice","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000007","input_cache_read":"0.000000025","discount":0},"name":"Qwen: Qwen3.6 35B A3B (cheapest-provider)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.6 35B A3B (quality)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b","provider":"Parasail","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005"},"name":"Qwen: Qwen3.6 35B A3B (normal)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-27b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.0000027","discount":0},"name":"Qwen: Qwen3.6 27B (fast)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b","provider":"Alibaba","variant":"fast","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000002","input_cache_read":"0.00000003","discount":0},"name":"Qwen: Qwen3.6 27B (cheapest-provider)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b","provider":"Chutes","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000032","discount":0},"name":"Qwen: Qwen3.6 27B (quality)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b","provider":"SiliconFlow","variant":"quality","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000032","completion":"0.00000325","input_cache_read":"0.00000003"},"name":"Qwen: Qwen3.6 27B (normal)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b","variant":"normal","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000150162","completion":"0.000003135","input_cache_read":"0.000000135","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0423 (fast)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro","provider":"SiliconFlow","variant":"fast","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174","discount":0.88},"name":"DeepSeek: DeepSeek V4 Pro 0423 (cheapest-provider)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-pro:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000348","input_cache_read":"0.0000001","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0423 (quality)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro","provider":"Parasail","variant":"quality","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"name":"DeepSeek: DeepSeek V4 Pro 0423 (normal)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048575,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000091","completion":"0.000000182","input_cache_read":"0.0000000182","discount":0.35},"name":"DeepSeek: DeepSeek V4 Flash 0423 (fast)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash","provider":"GMICloud","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.000000084","input_cache_read":"0.0000000084","discount":0.7},"name":"DeepSeek: DeepSeek V4 Flash 0423 (cheapest-provider)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.00000007","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0423 (quality)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash","provider":"Parasail","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000106","completion":"0.00000128","input_cache_read":"0.0000000106"},"name":"DeepSeek: DeepSeek V4 Flash 0423 (normal)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000048","completion":"0.0000018","input_cache_read":"0.000000096","discount":0},"name":"Xiaomi: MiMo-V2.5-Pro (fast)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro","provider":"DigitalOcean","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.5-pro:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro (cheapest-provider)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro","provider":"GMICloud","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro (quality)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro","provider":"GMICloud","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"name":"Xiaomi: MiMo-V2.5-Pro (normal)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000008","discount":0},"name":"Xiaomi: MiMo-V2.5 (fast)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5","provider":"Venice","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5 (cheapest-provider)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5","provider":"GMICloud","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.5:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5 (quality)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5","provider":"GMICloud","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"name":"Xiaomi: MiMo-V2.5 (normal)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2.6:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000057","completion":"0.0000024","input_cache_read":"0.000000114","discount":0},"name":"MoonshotAI: Kimi K2.6 (fast)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6","provider":"DigitalOcean","variant":"fast","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000465","completion":"0.00000245","input_cache_read":"0.0000000975","discount":0},"name":"MoonshotAI: Kimi K2.6 (cheapest-provider)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6","provider":"Inceptron","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000035","input_cache_read":"0.00000035","discount":0},"name":"MoonshotAI: Kimi K2.6 (quality)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6","provider":"Crusoe","variant":"quality","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.000004","input_cache_read":"0.00000016"},"name":"MoonshotAI: Kimi K2.6 (normal)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6","variant":"normal","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (fast)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7","provider":"Claude Platform on AWS","variant":"fast","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (cheapest-provider)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7 (normal)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7","variant":"normal","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.1 (fast)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1","provider":"Friendli","variant":"fast","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000966","completion":"0.000003036","input_cache_read":"0.0000001794","discount":0.31},"name":"Z.ai: GLM 5.1 (cheapest-provider)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000119","completion":"0.00000374","input_cache_read":"0.0000006","discount":0},"name":"Z.ai: GLM 5.1 (quality)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1","provider":"SiliconFlow","variant":"quality","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026"},"name":"Z.ai: GLM 5.1 (normal)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1","variant":"normal","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemma-4-26b-a4b-it:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 26B A4B  (fast)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it","provider":"CoreWeave","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.00000022","input_cache_read":"0.000000021","discount":0},"name":"Google: Gemma 4 26B A4B  (cheapest-provider)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 26B A4B  (quality)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it","provider":"CoreWeave","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 26B A4B  (normal)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000004","discount":0},"name":"Google: Gemma 4 31B (fast)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it","provider":"Friendli","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 31B (cheapest-provider)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000004","input_cache_read":"0.00000014","discount":0},"name":"Google: Gemma 4 31B (quality)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it","provider":"Crusoe","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 31B (normal)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"x-ai/grok-4.20-multi-agent:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (fast)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent","provider":"xAI","variant":"fast","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-multi-agent:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (cheapest-provider)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent","provider":"xAI","variant":"cheapest-provider","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-multi-agent-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (normal)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent","variant":"normal","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (fast)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20","provider":"xAI","variant":"fast","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (cheapest-provider)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20","provider":"xAI","variant":"cheapest-provider","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (normal)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20","variant":"normal","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.7:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042","discount":0.3},"name":"MiniMax: MiniMax M2.7 (fast)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7","provider":"GMICloud","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2.7:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042","discount":0.3},"name":"MiniMax: MiniMax M2.7 (cheapest-provider)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7","provider":"GMICloud","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2.7:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M2.7 (quality)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7","provider":"Minimax","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.7-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042"},"name":"MiniMax: MiniMax M2.7 (normal)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Small 4 (fast)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603","provider":"Mistral","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Small 4 (cheapest-provider)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603","provider":"Mistral","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Mistral: Mistral Small 4 (normal)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-9b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B (fast)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b","provider":"Darkbloom","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B (cheapest-provider)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000025","discount":0},"name":"Qwen: Qwen3.5-9B (quality)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b","provider":"Parasail","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000015"},"name":"Qwen: Qwen3.5-9B (normal)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (fast)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b","provider":"DeepInfra","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (cheapest-provider)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (quality)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b","provider":"Parasail","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004"},"name":"Qwen: Qwen3.5-35B-A3B (normal)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-27b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.0000026","discount":0},"name":"Qwen: Qwen3.5-27B (fast)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b","provider":"DeepInfra","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-27b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156","discount":0},"name":"Qwen: Qwen3.5-27B (cheapest-provider)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b","provider":"Alibaba","variant":"cheapest-provider","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-27b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000024","discount":0},"name":"Qwen: Qwen3.5-27B (quality)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b","provider":"Novita","variant":"quality","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-27b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"name":"Qwen: Qwen3.5-27B (normal)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-122b-a10b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (fast)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b","provider":"SiliconFlow","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (cheapest-provider)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b","provider":"SiliconFlow","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000032","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (quality)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b","provider":"Novita","variant":"quality","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"name":"Qwen: Qwen3.5-122B-A10B (normal)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (fast)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex","provider":"Azure","variant":"fast","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (cheapest-provider)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex","provider":"Azure","variant":"cheapest-provider","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex (normal)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex","variant":"normal","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (fast)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6","provider":"Claude Platform on AWS","variant":"fast","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (cheapest-provider)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6 (normal)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6","variant":"normal","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000036","input_cache_read":"0.0000003","discount":0},"name":"Qwen: Qwen3.5 397B A17B (fast)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b","provider":"Parasail","variant":"fast","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000039","completion":"0.00000234","discount":0},"name":"Qwen: Qwen3.5 397B A17B (cheapest-provider)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b","provider":"Alibaba","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000036","input_cache_read":"0.0000003","discount":0},"name":"Qwen: Qwen3.5 397B A17B (quality)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b","provider":"Parasail","variant":"quality","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.000003","input_cache_read":"0.00000022"},"name":"Qwen: Qwen3.5 397B A17B (normal)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b","variant":"normal","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000295","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M2.5 (fast)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5","provider":"AtlasCloud","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000095","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5 (cheapest-provider)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5","provider":"Venice","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2.5:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5 (quality)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5","provider":"Novita","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000108","input_cache_read":"0.000000027"},"name":"MiniMax: MiniMax M2.5 (normal)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5","variant":"normal","request_multiplier":3,"required_plans":null},{"id":"z-ai/glm-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.00000224","input_cache_read":"0.00000014","discount":0},"name":"Z.ai: GLM 5 (fast)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5","provider":"Baidu","variant":"fast","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012","discount":0.4},"name":"Z.ai: GLM 5 (cheapest-provider)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.00000255","input_cache_read":"0.0000002","discount":0},"name":"Z.ai: GLM 5 (quality)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5","provider":"SiliconFlow","variant":"quality","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"name":"Z.ai: GLM 5 (normal)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5","variant":"normal","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (fast)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6","provider":"Google","variant":"fast","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (cheapest-provider)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6 (normal)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6","variant":"normal","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-next:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 Coder Next (fast)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next","provider":"Novita","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder-next:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Coder Next (cheapest-provider)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next","provider":"Parasail","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-coder-next:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Coder Next (quality)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next","provider":"Parasail","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-coder-next-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Coder Next (normal)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"moonshotai/kimi-k2.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (fast)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5","provider":"SiliconFlow","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (cheapest-provider)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5","provider":"SiliconFlow","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (quality)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5","provider":"SiliconFlow","variant":"quality","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007"},"name":"MoonshotAI: Kimi K2.5 (normal)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5","variant":"normal","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-flash:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000605","completion":"0.0000004","discount":0},"name":"Z.ai: GLM 4.7 Flash (fast)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash","provider":"Cloudflare","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash (cheapest-provider)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash","provider":"Venice","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash (quality)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash","provider":"Novita","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000605","completion":"0.0000004"},"name":"Z.ai: GLM 4.7 Flash (normal)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (fast)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1","provider":"Novita","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (cheapest-provider)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1","provider":"Novita","variant":"cheapest-provider","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (quality)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1","provider":"Novita","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2.1 (normal)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-4.7:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","discount":0},"name":"Z.ai: GLM 4.7 (fast)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7","provider":"Google","variant":"fast","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.7 (cheapest-provider)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.00000054","completion":"0.00000198","input_cache_read":"0.000000099","discount":0.1},"name":"Z.ai: GLM 4.7 (quality)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7","provider":"Novita","variant":"quality","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.7 (normal)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7","variant":"normal","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-nano-30b-a3b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (fast)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b","provider":"Crusoe","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (cheapest-provider)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b","provider":"Crusoe","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (quality)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b","provider":"Crusoe","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (normal)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Ministral 3 14B 2512 (fast)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512","provider":"Mistral","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Ministral 3 14B 2512 (cheapest-provider)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512","provider":"Mistral","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"name":"Mistral: Ministral 3 14B 2512 (normal)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Ministral 3 8B 2512 (fast)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512","provider":"Mistral","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Ministral 3 8B 2512 (cheapest-provider)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512","provider":"Mistral","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"name":"Mistral: Ministral 3 8B 2512 (normal)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Ministral 3 3B 2512 (fast)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512","provider":"Mistral","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Ministral 3 3B 2512 (cheapest-provider)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512","provider":"Mistral","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"name":"Mistral: Ministral 3 3B 2512 (normal)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v3.2:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000056","completion":"0.00000168","discount":0},"name":"DeepSeek: DeepSeek V3.2 (fast)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2","provider":"Google","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000003096","input_cache_read":"0.0000000216","discount":0.28},"name":"DeepSeek: DeepSeek V3.2 (cheapest-provider)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2","provider":"GMICloud","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v3.2:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000259","completion":"0.00000042","input_cache_read":"0.000000135","discount":0},"name":"DeepSeek: DeepSeek V3.2 (quality)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2","provider":"SiliconFlow","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v3.2-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000028","completion":"0.00000042","input_cache_read":"0.000000028"},"name":"DeepSeek: DeepSeek V3.2 (normal)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-4.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.5 (fast)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5","provider":"Amazon Bedrock","variant":"fast","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.5 (cheapest-provider)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.5 (normal)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5","variant":"normal","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"mistralai/voxtral-small-24b-2507:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Voxtral Small 24B 2507 (fast)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507","provider":"Mistral","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/voxtral-small-24b-2507:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Voxtral Small 24B 2507 (cheapest-provider)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507","provider":"Mistral","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/voxtral-small-24b-2507-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"name":"Mistral: Voxtral Small 24B 2507 (normal)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (fast)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2","provider":"Minimax","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (cheapest-provider)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2","provider":"Minimax","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (quality)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2","provider":"Minimax","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012"},"name":"MiniMax: MiniMax M2 (normal)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-haiku-4.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002","discount":0},"name":"Anthropic: Claude Haiku 4.5 (fast)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5","provider":"Azure","variant":"fast","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002","discount":0},"name":"Anthropic: Claude Haiku 4.5 (cheapest-provider)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5","provider":"Azure","variant":"cheapest-provider","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"name":"Anthropic: Claude Haiku 4.5 (normal)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5","variant":"normal","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (fast)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct","provider":"Alibaba","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (cheapest-provider)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct","provider":"Alibaba","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000007","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (quality)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct","provider":"Novita","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"name":"Qwen: Qwen3 VL 30B A3B Instruct (normal)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.6:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000002","input_cache_read":"0.0000001","discount":0},"name":"Z.ai: GLM 4.6 (fast)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6","provider":"DeepInfra","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.6 (cheapest-provider)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6","provider":"Venice","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000022","input_cache_read":"0.00000011","discount":0},"name":"Z.ai: GLM 4.6 (quality)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6","provider":"Novita","variant":"quality","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008"},"name":"Z.ai: GLM 4.6 (normal)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (fast)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5","provider":"Claude Platform on AWS","variant":"fast","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (cheapest-provider)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5","provider":"Claude Platform on AWS","variant":"cheapest-provider","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (normal)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5","variant":"normal","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000104","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (fast)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct","provider":"Alibaba","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000088","input_cache_read":"0.00000011","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (cheapest-provider)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (quality)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct","provider":"Novita","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.0000019","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 VL 235B A22B Instruct (normal)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct","variant":"normal","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (fast)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct","provider":"Parasail","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000975","completion":"0.00000078","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (cheapest-provider)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct","provider":"Alibaba","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (quality)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct","provider":"Novita","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Next 80B A3B Instruct (normal)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013","discount":0},"name":"DeepSeek: DeepSeek V3.1 (fast)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1","provider":"DeepInfra","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013","discount":0},"name":"DeepSeek: DeepSeek V3.1 (cheapest-provider)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.000001","discount":0},"name":"DeepSeek: DeepSeek V3.1 (quality)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1","provider":"SiliconFlow","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"name":"DeepSeek: DeepSeek V3.1 (normal)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1","variant":"normal","request_multiplier":3,"required_plans":null},{"id":"mistralai/mistral-medium-3.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.0000022","input_cache_read":"0.000000044","discount":0},"name":"Mistral: Mistral Medium 3.1 (fast)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1","provider":"Mistral","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3.1 (cheapest-provider)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1","provider":"Mistral","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3.1 (normal)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125","discount":0},"name":"OpenAI: GPT-5 (fast)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5","provider":"OpenAI","variant":"fast","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125","discount":0},"name":"OpenAI: GPT-5 (cheapest-provider)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5","provider":"Azure","variant":"cheapest-provider","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5 (normal)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5","variant":"normal","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-oss-120b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","input_cache_read":"0.00000005","discount":0},"name":"OpenAI: gpt-oss-120b (fast)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b","provider":"Crusoe","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000017","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: gpt-oss-120b (cheapest-provider)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b","provider":"CoreWeave","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","input_cache_read":"0.00000005","discount":0},"name":"OpenAI: gpt-oss-120b (quality)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b","provider":"Crusoe","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000037","completion":"0.00000017"},"name":"OpenAI: gpt-oss-120b (normal)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000014","discount":0},"name":"OpenAI: gpt-oss-20b (fast)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b","provider":"DeepInfra","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009","discount":0},"name":"OpenAI: gpt-oss-20b (cheapest-provider)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b","provider":"Darkbloom","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000014","discount":0},"name":"OpenAI: gpt-oss-20b (quality)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b","provider":"DeepInfra","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"name":"OpenAI: gpt-oss-20b (normal)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (fast)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct","provider":"SiliconFlow","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":160000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000027","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (cheapest-provider)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct","provider":"Novita","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (quality)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct","provider":"SiliconFlow","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028"},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (normal)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (fast)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507","provider":"Alibaba","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004815","completion":"0.00000019305","discount":0.55},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (cheapest-provider)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507","provider":"StreamLake","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","discount":0},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (quality)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507","provider":"SiliconFlow","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (normal)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.5-air:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (fast)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air","provider":"Novita","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (cheapest-provider)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air","provider":"Novita","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (quality)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air","provider":"Novita","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"name":"Z.ai: GLM 4.5 Air (normal)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-thinking-2507:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (fast)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507","provider":"Alibaba","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (cheapest-provider)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507","provider":"Alibaba","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000003","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (quality)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507","provider":"Novita","variant":"quality","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023"},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (normal)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000018","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (fast)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder","provider":"Google","variant":"fast","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (cheapest-provider)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000038","completion":"0.00000155","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (quality)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder","provider":"Novita","variant":"quality","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 Coder 480B A35B (normal)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder","variant":"normal","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000008","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (fast)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507","provider":"Parasail","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (cheapest-provider)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507","provider":"GMICloud","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (quality)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507","provider":"GMICloud","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000055"},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (normal)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Mistral: Mistral Small 3.2 24B (fast)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct","provider":"Parasail","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000002","discount":0},"name":"Mistral: Mistral Small 3.2 24B (cheapest-provider)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Mistral: Mistral Small 3.2 24B (quality)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct","provider":"Parasail","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":256000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009375","completion":"0.00000025"},"name":"Mistral: Mistral Small 3.2 24B (normal)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-r1-0528:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035","discount":0},"name":"DeepSeek: R1 0528 (fast)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528","provider":"DeepInfra","variant":"fast","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035","discount":0},"name":"DeepSeek: R1 0528 (cheapest-provider)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000218","discount":0},"name":"DeepSeek: R1 0528 (quality)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528","provider":"SiliconFlow","variant":"quality","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"name":"DeepSeek: R1 0528 (normal)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528","variant":"normal","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3 (fast)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3","provider":"Mistral","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3 (cheapest-provider)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3","provider":"Mistral","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3 (normal)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-14b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B (fast)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b","provider":"NextBit","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B (cheapest-provider)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b","provider":"NextBit","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.00000024","discount":0},"name":"Qwen: Qwen3 14B (quality)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b","provider":"DeepInfra","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.00000024"},"name":"Qwen: Qwen3 14B (normal)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 32B (fast)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b","provider":"DeepInfra","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 32B (cheapest-provider)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000057","discount":0},"name":"Qwen: Qwen3 32B (quality)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b","provider":"SiliconFlow","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-32b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"name":"Qwen: Qwen3 32B (normal)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000022","completion":"0.0000088","web_search":"0.01","input_cache_read":"0.00000055","discount":0},"name":"OpenAI: GPT-4.1 (fast)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1","provider":"Azure","variant":"fast","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005","discount":0},"name":"OpenAI: GPT-4.1 (cheapest-provider)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1","provider":"Azure","variant":"cheapest-provider","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT-4.1 (normal)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1","variant":"normal","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001","discount":0},"name":"OpenAI: GPT-4.1 Mini (fast)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini","provider":"Azure","variant":"fast","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001","discount":0},"name":"OpenAI: GPT-4.1 Mini (cheapest-provider)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini","provider":"Azure","variant":"cheapest-provider","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"name":"OpenAI: GPT-4.1 Mini (normal)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini","variant":"normal","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-nano:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025","discount":0},"name":"OpenAI: GPT-4.1 Nano (fast)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano","provider":"OpenAI","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1-nano:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: GPT-4.1 Nano (cheapest-provider)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano","provider":"Azure","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1-nano-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"name":"OpenAI: GPT-4.1 Nano (normal)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-maverick:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":524288,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000035","completion":"0.000001","input_cache_read":"0.00000017","discount":0},"name":"Meta: Llama 4 Maverick (fast)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick","provider":"Parasail","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-4-maverick:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525","discount":0},"name":"Meta: Llama 4 Maverick (cheapest-provider)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick","provider":"DigitalOcean","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-maverick:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000085","discount":0},"name":"Meta: Llama 4 Maverick (quality)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick","provider":"Novita","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-4-maverick-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525"},"name":"Meta: Llama 4 Maverick (normal)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":327680,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Meta: Llama 4 Scout (fast)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout","provider":"DeepInfra","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-scout:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":327680,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Meta: Llama 4 Scout (cheapest-provider)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-scout:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.00000059","discount":0},"name":"Meta: Llama 4 Scout (quality)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout","provider":"Novita","variant":"quality","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":1310720,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Meta: Llama 4 Scout (normal)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":110000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Google: Gemma 3 27B (fast)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it","provider":"Nebius","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000016","discount":0},"name":"Google: Gemma 3 27B (cheapest-provider)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":98304,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.0000002","discount":0},"name":"Google: Gemma 3 27B (quality)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it","provider":"Novita","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000045","input_cache_read":"0.00000004"},"name":"Google: Gemma 3 27B (normal)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-saba:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Saba (fast)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba","provider":"Mistral","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-saba:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Saba (cheapest-provider)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba","provider":"Mistral","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-saba-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"name":"Mistral: Saba (normal)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000005","input_cache_read":"0.00000011","discount":0},"name":"Meta: Llama 3.3 70B Instruct (fast)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct","provider":"Parasail","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000032","discount":0},"name":"Meta: Llama 3.3 70B Instruct (cheapest-provider)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":12288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000135","completion":"0.0000004","discount":0},"name":"Meta: Llama 3.3 70B Instruct (quality)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct","provider":"Novita","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000005","input_cache_read":"0.00000011"},"name":"Meta: Llama 3.3 70B Instruct (normal)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-large-2407:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral Large 2407 (fast)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407","provider":"Mistral","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral Large 2407 (cheapest-provider)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407","provider":"Mistral","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large 2407 (normal)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"sao10k/l3-lunaris-8b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (fast)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b","provider":"Novita","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (cheapest-provider)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b","provider":"Parasail","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (quality)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b","provider":"Novita","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"name":"Sao10K: Llama 3 8B Lunaris (normal)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":32000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000152","completion":"0.000000287","discount":0},"name":"Meta: Llama 3.1 8B Instruct (fast)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct","provider":"Cloudflare","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000002","completion":"0.00000004","discount":0},"name":"Meta: Llama 3.1 8B Instruct (cheapest-provider)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct","provider":"DeepInfra","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.00000022","input_cache_read":"0.00000022","discount":0},"name":"Meta: Llama 3.1 8B Instruct (quality)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct","provider":"CoreWeave","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000008","input_cache_read":"0.000000025"},"name":"Meta: Llama 3.1 8B Instruct (normal)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000023","completion":"0.00000003","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Nemo (fast)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo","provider":"Io Net","variant":"fast","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000003","discount":0},"name":"Mistral: Mistral Nemo (cheapest-provider)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo","provider":"DekaLLM","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000023","completion":"0.00000003","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Nemo (quality)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo","provider":"Io Net","variant":"quality","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000019","completion":"0.00000003"},"name":"Mistral: Mistral Nemo (normal)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-mini:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.000000165","completion":"0.00000066","input_cache_read":"0.0000000825","discount":0},"name":"OpenAI: GPT-4o-mini (fast)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini","provider":"Azure","variant":"fast","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075","discount":0},"name":"OpenAI: GPT-4o-mini (cheapest-provider)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini","provider":"Azure","variant":"cheapest-provider","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini (normal)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini","variant":"normal","request_multiplier":2,"required_plans":null},{"id":"mistralai/mixtral-8x22b-instruct:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral: Mixtral 8x22B Instruct (fast)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct","provider":"Mistral","variant":"fast","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral: Mixtral 8x22B Instruct (cheapest-provider)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct","provider":"Mistral","variant":"cheapest-provider","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral: Mixtral 8x22B Instruct (normal)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct","variant":"normal","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"gryphe/mythomax-l2-13b:fast","object":"model","owned_by":"cv11","created":1791290914,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004","discount":0},"name":"MythoMax 13B (fast)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b","provider":"DeepInfra","variant":"fast","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b:cheapest-provider","object":"model","owned_by":"cv11","created":1791290914,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011","discount":0},"name":"MythoMax 13B (cheapest-provider)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b","provider":"Parasail","variant":"cheapest-provider","request_multiplier":1,"required_plans":null},{"id":"gryphe/mythomax-l2-13b:quality","object":"model","owned_by":"cv11","created":1791290914,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004","discount":0},"name":"MythoMax 13B (quality)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b","provider":"DeepInfra","variant":"quality","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b-normal","object":"model","owned_by":"cv11","created":1791290914,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"name":"MythoMax 13B (normal)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b","variant":"normal","request_multiplier":1,"required_plans":null},{"id":"lucidityai/synth-2.5-flash","object":"model","owned_by":"cv11","created":1790628514,"context_length":30000,"architecture":null,"pricing":{"prompt":"0.00000003","completion":"0.0000001"},"name":"LucidityAI: Synth 2.5 Flash Preview","description":null,"request_multiplier":0,"required_plans":null},{"id":"lucidityai/synth-2.5-pro","object":"model","owned_by":"cv11","created":1790628514,"context_length":30000,"architecture":null,"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"LucidityAI: Synth 2.5 Pro Preview","description":null,"request_multiplier":0,"required_plans":null},{"id":"lucidityai/synth-2.5-flash:free","object":"model","owned_by":"cv11","created":1791290919,"context_length":30000,"architecture":null,"pricing":{"prompt":"0.00000003","completion":"0.0000001"},"name":"LucidityAI: Synth 2.5 Flash Preview (free)","request_multiplier":0,"required_plans":null},{"id":"lucidityai/synth-2.5-pro:free","object":"model","owned_by":"cv11","created":1791290919,"context_length":30000,"architecture":null,"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"LucidityAI: Synth 2.5 Pro Preview (free)","request_multiplier":0,"required_plans":null},{"id":"composite/anthropic-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000005","completion":"0.000025"},"name":"Composite Anthropic Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/anthropic-router","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"composite/gemini-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000001","completion":"0.000006"},"name":"Composite Gemini Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/gemini-router","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"composite/gpt-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000001","completion":"0.000006"},"name":"Composite GPT Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/gpt-router","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"composite/planning-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"3.75e-7","completion":"0.00000225"},"name":"Composite Planning Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/planning-router","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"composite/execution-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"3.75e-7","completion":"0.00000225"},"name":"Composite Execution Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/execution-router","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"composite/subagent-router:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"1.19e-7","completion":"4e-7"},"name":"Composite Subagent Router (cheapThink)","router":true,"ensemble":false,"base_model":"composite/subagent-router","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"lucidityai/leviathan-2-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.0000019025","completion":"0.00000883"},"name":"Leviathan 2 Flash (cheapThink)","router":true,"ensemble":true,"base_model":"lucidityai/leviathan-2-flash","modifier":"cheapThink","request_multiplier":24,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"lucidityai/leviathan-2-base:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000005600000000000001","completion":"0.000024025"},"name":"Leviathan 2 Base (cheapThink)","router":true,"ensemble":true,"base_model":"lucidityai/leviathan-2-base","modifier":"cheapThink","request_multiplier":68,"required_plans":["pro","ultra","enterprise"]},{"id":"lucidityai/leviathan-2-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000009075","completion":"0.00004335000000000001"},"name":"Leviathan 2 Pro (cheapThink)","router":true,"ensemble":true,"base_model":"lucidityai/leviathan-2-pro","modifier":"cheapThink","request_multiplier":118,"required_plans":["ultra","enterprise"]},{"id":"lucidityai/leviathan-2-ultra:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":null,"pricing":{"prompt":"0.000019075000000000003","completion":"0.00009335"},"name":"Leviathan 2 Ultra (cheapThink)","router":true,"ensemble":true,"base_model":"lucidityai/leviathan-2-ultra","modifier":"cheapThink","request_multiplier":251,"required_plans":["enterprise"]},{"id":"z-ai/glm-5.3-prime-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.0000029049999999999997","completion":"0.0000045496000000000005"},"name":"Z.ai: GLM 5.3 Prime (Cheap) (cheapThink)","base_model":"z-ai/glm-5.3-prime-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flashx-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"pricing":{"prompt":"4.75e-7","completion":"6.462500000000001e-7"},"name":"Z.ai: GLM 5.3 FlashX (Cheap) (cheapThink)","base_model":"z-ai/glm-5.3-flashx-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.8-flash-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.8 Flash (Cheap) (cheapThink)","base_model":"google/gemini-3.8-flash-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":false},"request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.3-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"5.250000000000001e-7","completion":"6.824400000000001e-7"},"name":"Z.ai: GLM 5.3 (Cheap) (cheapThink)","base_model":"z-ai/glm-5.3-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"google/gemini-3.7-flash-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.7 Flash (Cheap) (cheapThink)","base_model":"google/gemini-3.7-flash-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-opus-5-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 5 (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-5-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.6-flash-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"4.800000000000001e-7","completion":"9e-7"},"name":"Google: Gemini 3.6 Flash (Cheap) (cheapThink)","base_model":"google/gemini-3.6-flash-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.2-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"6.675e-7","completion":"9.306e-7"},"name":"Z.ai: GLM 5.2 (Cheap) (cheapThink)","base_model":"z-ai/glm-5.2-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":false},"request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.8 (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.8-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"pricing":{"prompt":"8.550000000000001e-7","completion":"0.00000216"},"name":"Google: Gemini 3.5 Flash (Cheap) (cheapThink)","base_model":"google/gemini-3.5-flash-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.672,"baseline":"gemini-3.8-flash at its default reasoning effort","measured_on":"google/gemini-3.8-flash","extrapolated":true},"request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.7 (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.7-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000001071","completion":"0.0000015696119999999999"},"name":"Z.ai: GLM 5.1 (Cheap) (cheapThink)","base_model":"z-ai/glm-5.1-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.6 (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.6-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":false},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.5 (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.5-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flash-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"pricing":{"prompt":"2.55e-7","completion":"2.585e-7"},"name":"Z.ai: GLM 5.3 Flash (normal) (Cheap) (cheapThink)","base_model":"z-ai/glm-5.3-flash-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"1.7500000000000002e-7","completion":"0.000003619"},"name":"Z.ai: GLM 5.3 (normal) (Cheap) (cheapThink)","base_model":"z-ai/glm-5.3-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 5 (normal) (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-5-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"2.5700000000000004e-7","completion":"0.000006204"},"name":"Z.ai: GLM 5.2 (normal) (Cheap) (cheapThink)","base_model":"z-ai/glm-5.2-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.8 (normal) (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.8-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.7 (normal) (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.7-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000001505","completion":"0.0000022748000000000002"},"name":"Z.ai: GLM 5.1 (normal) (Cheap) (cheapThink)","base_model":"z-ai/glm-5.1-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.708,"baseline":"glm-5.2 at its default reasoning effort","measured_on":"z-ai/glm-5.2","extrapolated":true},"request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.6 (normal) (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.6-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-normal-cheap:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"pricing":{"prompt":"0.000005105","completion":"0.000010125"},"name":"Anthropic: Claude Opus 4.5 (normal) (Cheap) (cheapThink)","base_model":"anthropic/claude-opus-4.5-normal-cheap","modifier":"cheapThink","cheap":true,"redux":true,"savings":{"output_cost_delta":-0.595,"baseline":"the same model answering directly","measured_on":"anthropic/claude-opus-4.6","extrapolated":true},"request_multiplier":42,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"unbiased/pareto-26.10-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000032","input_cache_read":"0.00000003"},"name":"Pareto 26.10 Preview (cheapThink)","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks. This is a preview of the...","base_model":"unbiased/pareto-26.10-preview","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6.1-sol-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.00000005","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000001","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6.1 Sol Pro (cheapThink)","description":"GPT-6.1 Sol Pro is the same underlying model as [GPT-6.1 Sol](https://openrouter.ai/openai/gpt-6.1-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6.1-sol-pro","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6.1-sol:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.00000005","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000001","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6.1 Sol (cheapThink)","description":"GPT-6.1 Sol is an upgrade to GPT-6 Sol from OpenAI, positioned below the flagship GPT-6 Astra in the GPT-6 series. It is suited for agentic coding, computer use, document-heavy professional...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6.1-sol","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5.5 (cheapThink)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perceptron/perceptron-mk1.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":36864,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"name":"Perceptron: Perceptron Mk1.5 (cheapThink)","description":"Perceptron Mk1.5 is Perceptron's embodied reasoning model for physical agents. It accepts text, image, video, and audio input, and answers with text plus optional structured annotations: points, boxes, polygons, tracks,...","base_model":"perceptron/perceptron-mk1.5","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"fireworks/ember-1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","input_cache_read":"0.0000003"},"name":"Fireworks: Ember-1 (cheapThink)","description":"Ember-1 is a specialized reasoning model from Fireworks Research, built on [Kimi K3](https://openrouter.ai/moonshotai/kimi-k3). It is designed to make every token go further: it produces shorter reasoning traces, using roughly 40%...","base_model":"fireworks/ember-1","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-prime:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000028","completion":"0.0000088","input_cache_read":"0.00000056"},"name":"Z.ai: GLM 5.3 Prime (cheapThink)","description":"GLM-5.3-Prime is the high-speed variant of Z.ai's GLM-5.3, inheriting its full capabilities while delivering 1.5–2× the output throughput through inference acceleration. It supports text input and output with a 1M-token...","base_model":"z-ai/glm-5.3-prime","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-max-prime:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000005"},"name":"Qwen: Qwen3.8 Max Prime (cheapThink)","description":"Qwen3.8 Max Prime is a higher-throughput variant of Qwen3.8 Max from Alibaba's Qwen team, served as a separate SKU at a higher price point. It accepts text, image, and video...","base_model":"qwen/qwen3.8-max-prime","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.5-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"name":"AionLabs: Aion 3.5 Mini (cheapThink)","description":"Aion 3.5 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It is the smaller, lower-cost sibling of Aion 3.5 and uses...","base_model":"aion-labs/aion-3.5-mini","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"name":"AionLabs: Aion 3.5 (cheapThink)","description":"Aion 3.5 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each...","base_model":"aion-labs/aion-3.5","modifier":"cheapThink","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"upstage/solar-mini4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.000000005"},"name":"Upstage: Solar Mini 4 (cheapThink)","description":"Solar Mini 4 is Upstage's compact, cost-efficient language model, a 35B-parameter mixture-of-experts with 3B active parameters and a 524K context window. It is built for agentic use cases where response...","base_model":"upstage/solar-mini4","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"cohere/command-a-plus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":192000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000015","input_cache_read":"0.00000015"},"name":"Cohere: Command A+ (cheapThink)","description":"Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured...","base_model":"cohere/command-a-plus","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-6-luna-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna Pro (cheapThink)","description":"GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-luna-pro","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-luna:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna (cheapThink)","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-luna","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-sol-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol Pro (cheapThink)","description":"GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-sol-pro","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6-sol:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol (cheapThink)","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-sol","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"name":"Anthropic: Claude Opus 5.5 (cheapThink)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5","modifier":"cheapThink","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-pro-ultraspeed:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000435","completion":"0.0000087","input_cache_read":"0.000000036"},"name":"Xiaomi: MiMo-V2.6-Pro-UltraSpeed (cheapThink)","description":"MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x...","base_model":"xiaomi/mimo-v2.6-pro-ultraspeed","modifier":"cheapThink","request_multiplier":36,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (cheapThink)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (cheapThink)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"x-ai/grok-4.7:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (cheapThink)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-omni-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016"},"name":"Qwen: Qwen3.8 Omni Flash (cheapThink)","description":"Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,...","base_model":"qwen/qwen3.8-omni-flash","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"prism-ml/ternary-bonsai-2-27b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000005","input_cache_read":"0.0000000375"},"name":"PrismML: Ternary Bonsai 2 27B (cheapThink)","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","base_model":"prism-ml/ternary-bonsai-2-27b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flashx:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000037","completion":"0.00000125","input_cache_read":"0.00000009"},"name":"Z.ai: GLM 5.3 FlashX (cheapThink)","description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","base_model":"z-ai/glm-5.3-flashx","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"unbiased/pareto:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.0000075","input_cache_read":"0.00000025"},"name":"Pareto (cheapThink)","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks.","base_model":"unbiased/pareto","modifier":"cheapThink","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inference-net/schematron-v2-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000015","input_cache_read":"0.00000003"},"name":"Inference.net: Schematron V2 Turbo (cheapThink)","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","base_model":"inference-net/schematron-v2-turbo","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"inference-net/schematron-v2-small:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000023","input_cache_read":"0.00000005"},"name":"Inference.net: Schematron V2 Small (cheapThink)","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","base_model":"inference-net/schematron-v2-small","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"sakana/fugu-ultra-v2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"name":"Sakana: Fugu Ultra v2 (cheapThink)","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","base_model":"sakana/fugu-ultra-v2","modifier":"cheapThink","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"sakana/fugu-max:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.01","input_cache_read":"0.00000025"},"name":"Sakana: Fugu Max (cheapThink)","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","base_model":"sakana/fugu-max","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash-vl:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000021","completion":"0.0000000616","input_cache_read":"0.0000000042"},"name":"inclusionAI: Ling 3.0 Flash VL (cheapThink)","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","base_model":"inclusionai/ling-3.0-flash-vl","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (cheapThink)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"inception/mercury-2.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":260000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000004","completion":"0.00000015","input_cache_read":"0.000000004"},"name":"Inception: Mercury 2.5 (cheapThink)","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","base_model":"inception/mercury-2.5","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000025","completion":"0.0000001","input_cache_read":"0.0000000025"},"name":"Nex AGI: Nex-N2.5-Mini (cheapThink)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","base_model":"nex-agi/nex-n2.5-mini","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"Nex AGI: Nex-N2.5-Pro (cheapThink)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","base_model":"nex-agi/nex-n2.5-pro","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-astra:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}],"discount":0},"name":"OpenAI: GPT-6 Astra (cheapThink)","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-astra","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"openai/gpt-6-astra-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}],"discount":0},"name":"OpenAI: GPT-6 Astra Pro (cheapThink)","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-astra-pro","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-max-0902:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","input_cache_write":"0.0000025"},"name":"Qwen: Qwen3.8 Max (0902) (cheapThink)","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","base_model":"qwen/qwen3.8-max-0902","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta/muse-spark-1.3-contributor:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"name":"Meta: Muse Spark 1.3 Contributor (cheapThink)","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","base_model":"meta/muse-spark-1.3-contributor","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta/muse-spark-1.3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.3 (cheapThink)","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","base_model":"meta/muse-spark-1.3","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.8-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0.5},"name":"Google: Gemini 3.8 Flash (cheapThink)","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-3.8-flash","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5.1 (cheapThink)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"ibm-granite/granite-4.2-8b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"IBM: Granite 4.2 8B (cheapThink)","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","base_model":"ibm-granite/granite-4.2-8b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"tencent/hy4-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview (cheapThink)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash-fin:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.0000001232","input_cache_read":"0.0000000084"},"name":"inclusionAI: Ling 3.0 Flash Fin (cheapThink)","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","base_model":"inclusionai/ling-3.0-flash-fin","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.8-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016","input_cache_write":"0.0000002"},"name":"Qwen: Qwen3.8 Flash (cheapThink)","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","base_model":"qwen/qwen3.8-flash","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015","discount":0.5},"name":"Z.ai: GLM 5.3 Flash (cheapThink)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta/muse-spark-1.2-contributor:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"name":"Meta: Muse Spark 1.2 Contributor (cheapThink)","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","base_model":"meta/muse-spark-1.2-contributor","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (cheapThink)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"tencent/hy-mt2-1.8b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","stop","temperature"],"pricing":{"prompt":"0.000000044","completion":"0.000000177"},"name":"Tencent: Hy-MT2-1.8B (cheapThink)","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","base_model":"tencent/hy-mt2-1.8b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"tencent/hy-mt2-30b-a3b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"name":"Tencent: Hy-MT2-30B-A3B (cheapThink)","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","base_model":"tencent/hy-mt2-30b-a3b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"tencent/hy-mt2-7b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"name":"Tencent: Hy-MT2-7B (cheapThink)","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","base_model":"tencent/hy-mt2-7b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000132","input_cache_read":"0.000000078","discount":0.7},"name":"Z.ai: GLM 5.3 (cheapThink)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.00000053125","discount":0.25},"name":"Qwen: Qwen3.8 27B (cheapThink)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"google/gemini-3.7-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0.5},"name":"Google: Gemini 3.7 Flash (cheapThink)","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","provider":"Google","flex":true,"base_model":"google/gemini-3.7-flash","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-2-1-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000025"},"name":"ByteDance Seed: Seed 2.1 Turbo (cheapThink)","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","base_model":"bytedance-seed/seed-2-1-turbo","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"name":"Qwen: Qwen3.8 2.4T A95B (cheapThink)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-2.0-code:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.000001","completion":"0.000006"}]},"name":"ByteDance Seed: Seed-2.0-Code (cheapThink)","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","base_model":"bytedance-seed/seed-2.0-code","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}],"discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813 (cheapThink)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6 (cheapThink)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-lightning:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000049","completion":"0.00000014","input_cache_read":"0.0000000245","discount":0.3},"name":"NVIDIA: Nemotron 3.5 Lightning (cheapThink)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"sakana/sakana-namazu:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.00000095","completion":"0.000004","web_search":"0.007","input_cache_read":"0.00000015"},"name":"Sakana: Sakana Namazu (cheapThink)","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","base_model":"sakana/sakana-namazu","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"upstage/solar-pro4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000036","input_cache_read":"0.000000018"},"name":"Upstage: Solar Pro 4 (cheapThink)","description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","base_model":"upstage/solar-pro4","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta/muse-glimmer-30b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (cheapThink)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta/muse-spark-1.2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.2 (cheapThink)","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, and PDF documents, returns text, and offers a 1M-token context window....","base_model":"meta/muse-spark-1.2","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-flash-0731:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731 (cheapThink)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"thinkingmachines/inkling-small:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.0000012","input_cache_read":"0.0000001"},"name":"Thinking Machines: Inkling Small (cheapThink)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","base_model":"thinkingmachines/inkling-small","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.7-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000013","input_cache_read":"0.000000006","input_cache_write":"0.000000038","overrides":[{"min_prompt_tokens":32000,"prompt":"0.0000001","completion":"0.0000004","input_cache_read":"0.00000002","input_cache_write":"0.000000125"},{"min_prompt_tokens":256000,"prompt":"0.0000002","completion":"0.0000008","input_cache_read":"0.00000004","input_cache_write":"0.00000025"}]},"name":"Qwen: Qwen3.7 Flash (cheapThink)","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","base_model":"qwen/qwen3.7-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-opus-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5 (cheapThink)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"inclusionai/ling-3.0-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000021","completion":"0.000000063","input_cache_read":"0.0000000042"},"name":"inclusionAI: Ling 3.0 Flash (cheapThink)","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","base_model":"inclusionai/ling-3.0-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"poolside/laguna-s-2.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000009"},"name":"Poolside: Laguna S 2.1 (cheapThink)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","base_model":"poolside/laguna-s-2.1","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.6-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000208333333333333","discount":0},"name":"Google: Gemini 3.6 Flash (cheapThink)","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","provider":"Google","flex":true,"base_model":"google/gemini-3.6-flash","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash-lite:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.00000015","input_audio_cache":"0.000000015","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.000000015","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.5 Flash Lite (cheapThink)","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","provider":"Google","flex":true,"base_model":"google/gemini-3.5-flash-lite","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meituan/longcat-2.0:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048756,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.000000006"},"name":"Meituan: LongCat 2.0 (cheapThink)","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","base_model":"meituan/longcat-2.0","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"thinkingmachines/inkling:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.00000405","input_cache_read":"0.00000016"},"name":"Thinking Machines: Inkling (cheapThink)","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","base_model":"thinkingmachines/inkling","modifier":"cheapThink","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3 (cheapThink)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","modifier":"cheapThink","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta/muse-spark-1.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"name":"Meta: Muse Spark 1.1 (cheapThink)","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, and PDF documents and returns text, with a 1M-token context window....","base_model":"meta/muse-spark-1.1","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"kwaipilot/kat-coder-pro-v2.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000074","completion":"0.00000296","input_cache_read":"0.00000015"},"name":"Kwaipilot: KAT-Coder-Pro V2.5 (cheapThink)","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","base_model":"kwaipilot/kat-coder-pro-v2.5","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-luna-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna Pro (cheapThink)","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-luna-pro","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-luna:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna (cheapThink)","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-luna","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-terra-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra Pro (cheapThink)","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-terra-pro","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-terra:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra (cheapThink)","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-terra","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol Pro (cheapThink)","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-sol-pro","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol (cheapThink)","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-sol","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (cheapThink)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.0-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"name":"AionLabs: Aion-3.0-Mini (cheapThink)","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","base_model":"aion-labs/aion-3.0-mini","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-3.0:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"name":"AionLabs: Aion-3.0 (cheapThink)","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","base_model":"aion-labs/aion-3.0","modifier":"cheapThink","request_multiplier":25,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (cheapThink)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"poolside/laguna-xs-2.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000006","completion":"0.00000012","input_cache_read":"0.00000003"},"name":"Poolside: Laguna XS 2.1 (cheapThink)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","base_model":"poolside/laguna-xs-2.1","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-sonnet-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5 (cheapThink)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.1-flash-lite-image:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","temperature","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000015","image_output":"0.00003","web_search":"0.014"},"name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) (cheapThink)","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","base_model":"google/gemini-3.1-flash-lite-image","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"sakana/fugu-ultra:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high"]},"is_moderated":false,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"name":"Sakana: Fugu Ultra (cheapThink)","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","base_model":"sakana/fugu-ultra","modifier":"cheapThink","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-3.1-flash-image:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image) (cheapThink)","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","base_model":"google/gemini-3.1-flash-image","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3-pro-image:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","image_output":"0.00012","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"name":"Google: Nano Banana Pro (Gemini 3 Pro Image) (cheapThink)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","base_model":"google/gemini-3-pro-image","modifier":"cheapThink","request_multiplier":30,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005625","completion":"0.0000018","input_cache_read":"0.000000105","discount":0.25},"name":"Z.ai: GLM 5.2 (cheapThink)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code (cheapThink)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5 (cheapThink)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-content-safety:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002"},"name":"NVIDIA: Nemotron 3.5 Content Safety (cheapThink)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","base_model":"nvidia/nemotron-3.5-content-safety","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-ultra-550b-a55b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra (cheapThink)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.7-plus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000032","completion":"0.00000128","input_cache_read":"0.000000064","input_cache_write":"0.0000004","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000096","completion":"0.00000384","input_cache_read":"0.000000192","input_cache_write":"0.0000012"}]},"name":"Qwen: Qwen3.7 Plus (cheapThink)","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","base_model":"qwen/qwen3.7-plus","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.00000096","input_cache_read":"0.00000005","discount":0},"name":"MiniMax: MiniMax M3 (cheapThink)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"stepfun/step-3.7-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000115","input_cache_read":"0.00000004"},"name":"StepFun: Step 3.7 Flash (cheapThink)","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","base_model":"stepfun/step-3.7-flash","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"anthropic/claude-opus-4.8:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8 (cheapThink)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3.7-max:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001475","completion":"0.000004425","input_cache_read":"0.000000295","input_cache_write":"0.00000184375"},"name":"Qwen: Qwen3.7 Max (cheapThink)","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","base_model":"qwen/qwen3.7-max","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (cheapThink)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.5-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000075","completion":"0.0000045","image":"0.00000075","audio":"0.0000015","input_audio_cache":"0.00000015","web_search":"0.014","internal_reasoning":"0.0000045","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.5 Flash (cheapThink)","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","provider":"Google","flex":true,"base_model":"google/gemini-3.5-flash","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perceptron/perceptron-mk1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"name":"Perceptron: Perceptron Mk1 (cheapThink)","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","base_model":"perceptron/perceptron-mk1","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.1-flash-lite:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite (cheapThink)","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","provider":"Google","flex":true,"base_model":"google/gemini-3.1-flash-lite","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-chat-latest:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT Chat Latest (cheapThink)","description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","base_model":"openai/gpt-chat-latest","modifier":"cheapThink","request_multiplier":75,"required_plans":["pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (cheapThink)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"name":"Mistral: Mistral Medium 3.5 (cheapThink)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-plus-20260420:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000018","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000375","completion":"0.00000225","input_cache_write":"0.00000046875"}]},"name":"Qwen: Qwen3.5 Plus 2026-04-20 (cheapThink)","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","base_model":"qwen/qwen3.5-plus-20260420","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.000001125","input_cache_write":"0.000000234375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000075","completion":"0.000003","input_cache_write":"0.0000009375"}]},"name":"Qwen: Qwen3.6 Flash (cheapThink)","description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","base_model":"qwen/qwen3.6-flash","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000007","input_cache_read":"0.000000025","discount":0},"name":"Qwen: Qwen3.6 35B A3B (cheapThink)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.6-max-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001027","completion":"0.000006162","input_cache_write":"0.00000128375","overrides":[{"min_prompt_tokens":128000,"prompt":"0.00000158","completion":"0.00000948","input_cache_write":"0.000001975"}]},"name":"Qwen: Qwen3.6 Max Preview (cheapThink)","description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","base_model":"qwen/qwen3.6-max-preview","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000002","input_cache_read":"0.00000003","discount":0},"name":"Qwen: Qwen3.6 27B (cheapThink)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.5-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}],"discount":0},"name":"OpenAI: GPT-5.5 Pro (cheapThink)","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.5-pro","modifier":"cheapThink","request_multiplier":225,"required_plans":["enterprise"]},{"id":"openai/gpt-5.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}],"discount":0},"name":"OpenAI: GPT-5.5 (cheapThink)","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.5","modifier":"cheapThink","request_multiplier":38,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"name":"DeepSeek: DeepSeek V4 Pro 0423 (cheapThink)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.000000084","input_cache_read":"0.0000000084","discount":0.7},"name":"DeepSeek: DeepSeek V4 Flash 0423 (cheapThink)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"tencent/hy3-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.0000006","input_cache_read":"0.00000006"},"name":"Tencent: Hy3 preview (cheapThink)","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","base_model":"tencent/hy3-preview","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro (cheapThink)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5 (cheapThink)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.4-image-2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":272000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","top_logprobs","verbosity"],"pricing":{"prompt":"0.000008","completion":"0.000015","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"name":"OpenAI: GPT-5.4 Image 2 (cheapThink)","description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","base_model":"openai/gpt-5.4-image-2","modifier":"cheapThink","request_multiplier":65,"required_plans":["pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000465","completion":"0.00000245","input_cache_read":"0.0000000975","discount":0},"name":"MoonshotAI: Kimi K2.6 (cheapThink)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7 (cheapThink)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000966","completion":"0.000003036","input_cache_read":"0.0000001794","discount":0.31},"name":"Z.ai: GLM 5.1 (cheapThink)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemma-4-26b-a4b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.00000022","input_cache_read":"0.000000021","discount":0},"name":"Google: Gemma 4 26B A4B  (cheapThink)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 31B (cheapThink)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.6-plus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000325","completion":"0.00000195","input_cache_write":"0.00000040625","overrides":[{"min_prompt_tokens":256000,"prompt":"0.0000013","completion":"0.0000039","input_cache_write":"0.000001625"}]},"name":"Qwen: Qwen3.6 Plus (cheapThink)","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","base_model":"qwen/qwen3.6-plus","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5v-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"name":"Z.ai: GLM 5V Turbo (cheapThink)","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","base_model":"z-ai/glm-5v-turbo","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"arcee-ai/trinity-large-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000008","input_cache_read":"0.00000006"},"name":"Arcee AI: Trinity Large Thinking (cheapThink)","description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","base_model":"arcee-ai/trinity-large-thinking","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"x-ai/grok-4.20-multi-agent:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (cheapThink)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (cheapThink)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"rekaai/reka-edge:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":16384,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001"},"name":"Reka Edge (cheapThink)","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","base_model":"rekaai/reka-edge","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2.7:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042"},"name":"MiniMax: MiniMax M2.7 (cheapThink)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-nano:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.000000625","web_search":"0.01","input_cache_read":"0.00000001","discount":0},"name":"OpenAI: GPT-5.4 Nano (cheapThink)","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4-nano","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000375","completion":"0.00000225","web_search":"0.01","input_cache_read":"0.0000000375","discount":0},"name":"OpenAI: GPT-5.4 Mini (cheapThink)","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4-mini","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-small-2603:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Mistral: Mistral Small 4 (cheapThink)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"name":"Z.ai: GLM 5 Turbo (cheapThink)","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","base_model":"z-ai/glm-5-turbo","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-super-120b-a12b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000045"},"name":"NVIDIA: Nemotron 3 Super (cheapThink)","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","base_model":"nvidia/nemotron-3-super-120b-a12b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"bytedance-seed/seed-2.0-lite:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"name":"ByteDance Seed: Seed-2.0-Lite (cheapThink)","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","base_model":"bytedance-seed/seed-2.0-lite","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-9b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B (cheapThink)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.4-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}],"discount":0},"name":"OpenAI: GPT-5.4 Pro (cheapThink)","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4-pro","modifier":"cheapThink","request_multiplier":225,"required_plans":["enterprise"]},{"id":"openai/gpt-5.4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000025","completion":"0.00001125","input_cache_read":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.4 (cheapThink)","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4","modifier":"cheapThink","request_multiplier":19,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inception/mercury-2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"name":"Inception: Mercury 2 (cheapThink)","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","base_model":"inception/mercury-2","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.1-flash-lite-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite Preview (cheapThink)","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-3.1-flash-lite-preview","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"bytedance-seed/seed-2.0-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000002","completion":"0.0000008"}]},"name":"ByteDance Seed: Seed-2.0-Mini (cheapThink)","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","base_model":"bytedance-seed/seed-2.0-mini","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.1-flash-image-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview) (cheapThink)","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","base_model":"google/gemini-3.1-flash-image-preview","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-35b-a3b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004"},"name":"Qwen: Qwen3.5-35B-A3B (cheapThink)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-27b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"name":"Qwen: Qwen3.5-27B (cheapThink)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-122b-a10b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"name":"Qwen: Qwen3.5-122B-A10B (cheapThink)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-flash-02-23:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000065","completion":"0.00000026"},"name":"Qwen: Qwen3.5-Flash (cheapThink)","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","base_model":"qwen/qwen3.5-flash-02-23","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemini-3.1-pro-preview-customtools:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","audio","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018","audio":"0.000004","input_audio_cache":"0.0000004","input_cache_read":"0.0000004"}]},"name":"Google: Gemini 3.1 Pro Preview Custom Tools (cheapThink)","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","base_model":"google/gemini-3.1-pro-preview-customtools","modifier":"cheapThink","request_multiplier":30,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex (cheapThink)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-2.0:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000016","input_cache_read":"0.0000002"},"name":"AionLabs: Aion-2.0 (cheapThink)","description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","base_model":"aion-labs/aion-2.0","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3.1-pro-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["audio","file","image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000006","image":"0.000001","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.000006","input_cache_read":"0.0000001","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000009","audio":"0.000002","input_audio_cache":"0.0000002","input_cache_read":"0.0000002"}],"discount":0},"name":"Google: Gemini 3.1 Pro Preview (cheapThink)","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","provider":"Google","flex":true,"base_model":"google/gemini-3.1-pro-preview","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6 (cheapThink)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-plus-02-15:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000156","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000325","completion":"0.00000195"}]},"name":"Qwen: Qwen3.5 Plus 2026-02-15 (cheapThink)","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","base_model":"qwen/qwen3.5-plus-02-15","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-397b-a17b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000039","completion":"0.00000234","input_cache_read":"0.00000022","discount":0},"name":"Qwen: Qwen3.5 397B A17B (cheapThink)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000095","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5 (cheapThink)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"z-ai/glm-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"name":"Z.ai: GLM 5 (cheapThink)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-max-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000078","completion":"0.0000039","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975"}]},"name":"Qwen: Qwen3 Max Thinking (cheapThink)","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","base_model":"qwen/qwen3-max-thinking","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6 (cheapThink)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-next:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Coder Next (cheapThink)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"stepfun/step-3.5-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"StepFun: Step 3.5 Flash (cheapThink)","description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","base_model":"stepfun/step-3.5-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007"},"name":"MoonshotAI: Kimi K2.5 (cheapThink)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"upstage/solar-pro-3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Upstage: Solar Pro 3 (cheapThink)","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","base_model":"upstage/solar-pro-3","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2-her:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2-her (cheapThink)","description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","base_model":"minimax/minimax-m2-her","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"writer/palmyra-x5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1040000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.000006"},"name":"Writer: Palmyra X5 (cheapThink)","description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","base_model":"writer/palmyra-x5","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-audio:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","audio"],"output_modalities":["text","audio"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001","audio":"0.000032","audio_output":"0.000064"},"name":"OpenAI: GPT Audio (cheapThink)","description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","base_model":"openai/gpt-audio","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-audio-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","audio"],"output_modalities":["text","audio"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000024","audio":"0.0000006","audio_output":"0.0000024"},"name":"OpenAI: GPT Audio Mini (cheapThink)","description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","base_model":"openai/gpt-audio-mini","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash (cheapThink)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.2-codex:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.2-Codex (cheapThink)","description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","base_model":"openai/gpt-5.2-codex","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"bytedance-seed/seed-1.6-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000001","completion":"0.0000008"}]},"name":"ByteDance Seed: Seed 1.6 Flash (cheapThink)","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","base_model":"bytedance-seed/seed-1.6-flash","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"bytedance-seed/seed-1.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"name":"ByteDance Seed: Seed 1.6 (cheapThink)","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","base_model":"bytedance-seed/seed-1.6","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2.1 (cheapThink)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-4.7:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.7 (cheapThink)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-3-flash-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3 Flash Preview (cheapThink)","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","provider":"Google","flex":true,"base_model":"google/gemini-3-flash-preview","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (cheapThink)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.2-chat:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_completion_tokens","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.2 Chat (cheapThink)","description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","base_model":"openai/gpt-5.2-chat","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.2-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000021","completion":"0.000168","web_search":"0.01"},"name":"OpenAI: GPT-5.2 Pro (cheapThink)","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","base_model":"openai/gpt-5.2-pro","modifier":"cheapThink","request_multiplier":385,"required_plans":["enterprise"]},{"id":"openai/gpt-5.2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000875","completion":"0.000007","web_search":"0.01","input_cache_read":"0.0000000875","discount":0},"name":"OpenAI: GPT-5.2 (cheapThink)","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.2","modifier":"cheapThink","request_multiplier":16,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/devstral-2512:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Devstral 2 2512 (cheapThink)","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","base_model":"mistralai/devstral-2512","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"relace/relace-search:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000003"},"name":"Relace: Relace Search (cheapThink)","description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","base_model":"relace/relace-search","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6v:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000005"},"name":"Z.ai: GLM 4.6V (cheapThink)","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","base_model":"z-ai/glm-4.6v","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"openai/gpt-5.1-codex-max:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5.1-Codex-Max (cheapThink)","description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","base_model":"openai/gpt-5.1-codex-max","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"amazon/nova-2-lite-v1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000025"},"name":"Amazon: Nova 2 Lite (cheapThink)","description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","base_model":"amazon/nova-2-lite-v1","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/ministral-14b-2512:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"name":"Mistral: Ministral 3 14B 2512 (cheapThink)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"name":"Mistral: Ministral 3 8B 2512 (cheapThink)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"name":"Mistral: Ministral 3 3B 2512 (cheapThink)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-large-2512:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000015","input_cache_read":"0.00000005"},"name":"Mistral: Mistral Large 3 2512 (cheapThink)","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","base_model":"mistralai/mistral-large-2512","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000003096","input_cache_read":"0.0000000216","discount":0.28},"name":"DeepSeek: DeepSeek V3.2 (cheapThink)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-4.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.5 (cheapThink)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-3-pro-image-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000006","image":"0.000001","image_output":"0.00006","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.000006","input_cache_read":"0.0000001","input_cache_write":"0.0000001875","discount":0},"name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview) (cheapThink)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-3-pro-image-preview","modifier":"cheapThink","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000625","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000000625","discount":0},"name":"OpenAI: GPT-5.1 (cheapThink)","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.1","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-codex:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.00000013"},"name":"OpenAI: GPT-5.1-Codex (cheapThink)","description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","base_model":"openai/gpt-5.1-codex","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-codex-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.00000003"},"name":"OpenAI: GPT-5.1-Codex-Mini (cheapThink)","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","base_model":"openai/gpt-5.1-codex-mini","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"name":"MoonshotAI: Kimi K2 Thinking (cheapThink)","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","base_model":"moonshotai/kimi-k2-thinking","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"amazon/nova-premier-v1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.0000125","input_cache_read":"0.000000625"},"name":"Amazon: Nova Premier 1.0 (cheapThink)","description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","base_model":"amazon/nova-premier-v1","modifier":"cheapThink","request_multiplier":33,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-pro-search:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.018"},"name":"Perplexity: Sonar Pro Search (cheapThink)","description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","base_model":"perplexity/sonar-pro-search","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/voxtral-small-24b-2507:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"name":"Mistral: Voxtral Small 24B 2507 (cheapThink)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-safeguard-20b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000003","input_cache_read":"0.0000000375"},"name":"OpenAI: gpt-oss-safeguard-20b (cheapThink)","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","base_model":"openai/gpt-oss-safeguard-20b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (cheapThink)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-vl-32b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000104","completion":"0.000000416"},"name":"Qwen: Qwen3 VL 32B Instruct (cheapThink)","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","base_model":"qwen/qwen3-vl-32b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"ibm-granite/granite-4.0-h-micro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000017","completion":"0.000000112"},"name":"IBM: Granite 4.0 Micro (cheapThink)","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","base_model":"ibm-granite/granite-4.0-h-micro","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5-image-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"pricing":{"prompt":"0.0000025","completion":"0.000002","image_output":"0.000008","web_search":"0.01","input_cache_read":"0.00000025"},"name":"OpenAI: GPT-5 Image Mini (cheapThink)","description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","base_model":"openai/gpt-5-image-mini","modifier":"cheapThink","request_multiplier":16,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"name":"Anthropic: Claude Haiku 4.5 (cheapThink)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-8b-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.0000021"},"name":"Qwen: Qwen3 VL 8B Thinking (cheapThink)","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","base_model":"qwen/qwen3-vl-8b-thinking","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-vl-8b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"name":"Qwen: Qwen3 VL 8B Instruct (cheapThink)","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","base_model":"qwen/qwen3-vl-8b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5-image:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["image","text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00001","image_output":"0.00004","web_search":"0.01","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-5 Image (cheapThink)","description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","base_model":"openai/gpt-5-image","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-2.5-flash-image:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["image","text"],"output_modalities":["image","text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","image_output":"0.00003","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"name":"Google: Nano Banana (Gemini 2.5 Flash Image) (cheapThink)","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","base_model":"google/gemini-2.5-flash-image","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"name":"Qwen: Qwen3 VL 30B A3B Thinking (cheapThink)","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","base_model":"qwen/qwen3-vl-30b-a3b-thinking","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (cheapThink)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000015","completion":"0.00012","web_search":"0.01"},"name":"OpenAI: GPT-5 Pro (cheapThink)","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","base_model":"openai/gpt-5-pro","modifier":"cheapThink","request_multiplier":275,"required_plans":["enterprise"]},{"id":"z-ai/glm-4.6:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008"},"name":"Z.ai: GLM 4.6 (cheapThink)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (cheapThink)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2-exp:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000041"},"name":"DeepSeek: DeepSeek V3.2 Exp (cheapThink)","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2-exp","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"thedrummer/cydonia-24b-v4.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000005","input_cache_read":"0.00000015"},"name":"TheDrummer: Cydonia 24B V4.1 (cheapThink)","description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","base_model":"thedrummer/cydonia-24b-v4.1","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"relace/relace-apply-3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","seed","stop"],"pricing":{"prompt":"0.00000085","completion":"0.00000125"},"name":"Relace: Relace Apply 3 (cheapThink)","description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","base_model":"relace/relace-apply-3","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000004"},"name":"Qwen: Qwen3 VL 235B A22B Thinking (cheapThink)","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","base_model":"qwen/qwen3-vl-235b-a22b-thinking","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000088","input_cache_read":"0.00000011","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (cheapThink)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-max:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000078","completion":"0.0000039","input_cache_read":"0.000000156","input_cache_write":"0.000000975","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078","input_cache_read":"0.000000312","input_cache_write":"0.00000195"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"name":"Qwen: Qwen3 Max (cheapThink)","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","base_model":"qwen/qwen3-max","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-plus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000325","input_cache_read":"0.00000013","input_cache_write":"0.0000008125","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000117","completion":"0.00000585","input_cache_read":"0.000000234","input_cache_write":"0.0000014625"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"name":"Qwen: Qwen3 Coder Plus (cheapThink)","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","base_model":"qwen/qwen3-coder-plus","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.1-terminus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.000001"},"name":"DeepSeek: DeepSeek V3.1 Terminus (cheapThink)","description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","base_model":"deepseek/deepseek-v3.1-terminus","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.000000975","input_cache_read":"0.000000039","input_cache_write":"0.00000024375","overrides":[{"min_prompt_tokens":32000,"prompt":"0.000000325","completion":"0.000001625","input_cache_read":"0.000000065","input_cache_write":"0.00000040625"},{"min_prompt_tokens":128000,"prompt":"0.00000052","completion":"0.0000026","input_cache_read":"0.000000104","input_cache_write":"0.00000065"}]},"name":"Qwen: Qwen3 Coder Flash (cheapThink)","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","base_model":"qwen/qwen3-coder-flash","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-thinking:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000012"},"name":"Qwen: Qwen3 Next 80B A3B Thinking (cheapThink)","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","base_model":"qwen/qwen3-next-80b-a3b-thinking","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000975","completion":"0.00000078","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (cheapThink)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen-plus-2025-07-28:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000078","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234"}]},"name":"Qwen: Qwen Plus 0728 (cheapThink)","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","base_model":"qwen/qwen-plus-2025-07-28","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"moonshotai/kimi-k2-0905:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"name":"MoonshotAI: Kimi K2 0905 (cheapThink)","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","base_model":"moonshotai/kimi-k2-0905","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-30b-a3b-thinking-2507:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":81920,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"name":"Qwen: Qwen3 30B A3B Thinking 2507 (cheapThink)","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","base_model":"qwen/qwen3-30b-a3b-thinking-2507","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-4-405b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000003"},"name":"Nous: Hermes 4 405B (cheapThink)","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","base_model":"nousresearch/hermes-4-405b","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-chat-v3.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"name":"DeepSeek: DeepSeek V3.1 (cheapThink)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"mistralai/mistral-medium-3.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3.1 (cheapThink)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.5v:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000018","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.5V (cheapThink)","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","base_model":"z-ai/glm-4.5v","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5 (cheapThink)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000125","completion":"0.000001","web_search":"0.01","input_cache_read":"0.0000000125","discount":0},"name":"OpenAI: GPT-5 Mini (cheapThink)","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5-mini","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5-nano:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000025","completion":"0.0000002","web_search":"0.01","input_cache_read":"0.0000000025","discount":0},"name":"OpenAI: GPT-5 Nano (cheapThink)","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5-nano","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000017","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: gpt-oss-120b (cheapThink)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"name":"OpenAI: gpt-oss-20b (cheapThink)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"anthropic/claude-opus-4.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000015","completion":"0.000075","web_search":"0.01","input_cache_read":"0.0000015","input_cache_write":"0.00001875","input_cache_write_1h":"0.00003"},"name":"Anthropic: Claude Opus 4.1 (cheapThink)","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","base_model":"anthropic/claude-opus-4.1","modifier":"cheapThink","request_multiplier":200,"required_plans":["ultra","enterprise"]},{"id":"mistralai/codestral-2508:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000003"},"name":"Mistral: Codestral 2508 (cheapThink)","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","base_model":"mistralai/codestral-2508","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000027","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (cheapThink)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004815","completion":"0.00000019305","discount":0.55},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (cheapThink)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.5:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.5 (cheapThink)","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","base_model":"z-ai/glm-4.5","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.5-air:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"name":"Z.ai: GLM 4.5 Air (cheapThink)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-thinking-2507:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023"},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (cheapThink)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 Coder 480B A35B (cheapThink)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"bytedance/ui-tars-1.5-7b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002","input_cache_read":"0.0000001"},"name":"ByteDance: UI-TARS 7B  (cheapThink)","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","base_model":"bytedance/ui-tars-1.5-7b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemini-2.5-flash-lite:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","image":"0.00000005","audio":"0.00000015","input_audio_cache":"0.000000015","web_search":"0.014","internal_reasoning":"0.0000002","input_cache_read":"0.000000005","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 2.5 Flash Lite (cheapThink)","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-2.5-flash-lite","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (cheapThink)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000057","completion":"0.0000023"},"name":"MoonshotAI: Kimi K2 0711 (cheapThink)","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","base_model":"moonshotai/kimi-k2","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000009"},"name":"Venice: Uncensored (cheapThink)","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","base_model":"cognitivecomputations/dolphin-mistral-24b-venice-edition","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"tencent/hunyuan-a13b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000057"},"name":"Tencent: Hunyuan A13B Instruct (cheapThink)","description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","base_model":"tencent/hunyuan-a13b-instruct","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"morph/morph-v3-large:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["logprobs","max_tokens","response_format","stop","structured_outputs","temperature","top_logprobs"],"pricing":{"prompt":"0.0000009","completion":"0.0000019"},"name":"Morph: Morph V3 Large (cheapThink)","description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","base_model":"morph/morph-v3-large","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"morph/morph-v3-fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":81920,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","stop","temperature"],"pricing":{"prompt":"0.0000008","completion":"0.0000012"},"name":"Morph: Morph V3 Fast (cheapThink)","description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","base_model":"morph/morph-v3-fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"baidu/ernie-4.5-vl-424b-a47b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":123000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000125"},"name":"Baidu: ERNIE 4.5 VL 424B A47B  (cheapThink)","description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","base_model":"baidu/ernie-4.5-vl-424b-a47b","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000002","discount":0},"name":"Mistral: Mistral Small 3.2 24B (cheapThink)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000022"},"name":"MiniMax: MiniMax M1 (cheapThink)","description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","base_model":"minimax/minimax-m1","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"google/gemini-2.5-flash:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["file","image","text","audio","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.000000015","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 2.5 Flash (cheapThink)","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-2.5-flash","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"google/gemini-2.5-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000005","image":"0.000000625","audio":"0.000000625","input_audio_cache":"0.0000000625","web_search":"0.014","internal_reasoning":"0.000005","input_cache_read":"0.0000000625","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.00000125","completion":"0.0000075","audio":"0.00000125","input_audio_cache":"0.000000125","input_cache_read":"0.000000125"}],"discount":0},"name":"Google: Gemini 2.5 Pro (cheapThink)","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-2.5-pro","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o3-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","file","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.00002","completion":"0.00008","web_search":"0.01"},"name":"OpenAI: o3 Pro (cheapThink)","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","base_model":"openai/o3-pro","modifier":"cheapThink","request_multiplier":233,"required_plans":["enterprise"]},{"id":"google/gemini-2.5-pro-preview:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["file","image","text","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000005","image":"0.000000625","audio":"0.000000625","input_audio_cache":"0.0000000625","web_search":"0.014","internal_reasoning":"0.000005","input_cache_read":"0.0000000625","input_cache_write":"0.0000001875","overrides":[{"min_prompt_tokens":200000,"prompt":"0.00000125","completion":"0.0000075","audio":"0.00000125","input_audio_cache":"0.000000125","input_cache_read":"0.000000125"}],"discount":0},"name":"Google: Gemini 2.5 Pro Preview 06-05 (cheapThink)","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-2.5-pro-preview","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"name":"DeepSeek: R1 0528 (cheapThink)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4 (cheapThink)","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","base_model":"anthropic/claude-sonnet-4","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3 (cheapThink)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"meta-llama/llama-guard-4-12b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"name":"Meta: Llama Guard 4 12B (cheapThink)","description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","base_model":"meta-llama/llama-guard-4-12b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000005"},"name":"Qwen: Qwen3 30B A3B (cheapThink)","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","base_model":"qwen/qwen3-30b-a3b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-8b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"name":"Qwen: Qwen3 8B (cheapThink)","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","base_model":"qwen/qwen3-8b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B (cheapThink)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"name":"Qwen: Qwen3 32B (cheapThink)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000455","completion":"0.00000182"},"name":"Qwen: Qwen3 235B A22B (cheapThink)","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","base_model":"qwen/qwen3-235b-a22b","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/o4-mini-high:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"name":"OpenAI: o4 Mini High (cheapThink)","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","base_model":"openai/o4-mini-high","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: o3 (cheapThink)","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","base_model":"openai/o3","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/o4-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"name":"OpenAI: o4 Mini (cheapThink)","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","base_model":"openai/o4-mini","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT-4.1 (cheapThink)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"name":"OpenAI: GPT-4.1 Mini (cheapThink)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-nano:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"name":"OpenAI: GPT-4.1 Nano (cheapThink)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-maverick:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525"},"name":"Meta: Llama 4 Maverick (cheapThink)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1310720,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Meta: Llama 4 Scout (cheapThink)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-chat-v3-0324:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000029","completion":"0.00000114","input_cache_read":"0.00000011"},"name":"DeepSeek: DeepSeek V3 0324 (cheapThink)","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","base_model":"deepseek/deepseek-chat-v3-0324","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"openai/o1-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"pricing":{"prompt":"0.00015","completion":"0.0006","web_search":"0.01"},"name":"OpenAI: o1-pro (cheapThink)","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","base_model":"openai/o1-pro","modifier":"cheapThink","request_multiplier":1750,"required_plans":["enterprise"]},{"id":"mistralai/mistral-small-3.1-24b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000351","completion":"0.000000555"},"name":"Mistral: Mistral Small 3.1 24B (cheapThink)","description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","base_model":"mistralai/mistral-small-3.1-24b-instruct","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"google/gemma-3-4b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000001"},"name":"Google: Gemma 3 4B (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-4b-it","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-12b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000015"},"name":"Google: Gemma 3 12B (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-12b-it","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"cohere/command-a:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"name":"Cohere: Command A (cheapThink)","description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","base_model":"cohere/command-a","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"rekaai/reka-flash-3:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"Reka Flash 3 (cheapThink)","description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","base_model":"rekaai/reka-flash-3","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000016","input_cache_read":"0.00000004","discount":0},"name":"Google: Gemma 3 27B (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"thedrummer/skyfall-36b-v2:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000008","input_cache_read":"0.00000025"},"name":"TheDrummer: Skyfall 36B V2 (cheapThink)","description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","base_model":"thedrummer/skyfall-36b-v2","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"perplexity/sonar-reasoning-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005"},"name":"Perplexity: Sonar Reasoning Pro (cheapThink)","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","base_model":"perplexity/sonar-reasoning-pro","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-pro:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005"},"name":"Perplexity: Sonar Pro (cheapThink)","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","base_model":"perplexity/sonar-pro","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"perplexity/sonar-deep-research:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005","internal_reasoning":"0.000003"},"name":"Perplexity: Sonar Deep Research (cheapThink)","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","base_model":"perplexity/sonar-deep-research","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-saba:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"name":"Mistral: Saba (cheapThink)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/o3-mini-high:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"name":"OpenAI: o3 Mini High (cheapThink)","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","base_model":"openai/o3-mini-high","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"aion-labs/aion-rp-llama-3.1-8b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000016"},"name":"AionLabs: Aion-RP 1.0 (8B) (cheapThink)","description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","base_model":"aion-labs/aion-rp-llama-3.1-8b","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen2.5-vl-72b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.000001","input_cache_read":"0.0000004"},"name":"Qwen: Qwen2.5 VL 72B Instruct (cheapThink)","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","base_model":"qwen/qwen2.5-vl-72b-instruct","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-plus:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_read":"0.000000052","input_cache_write":"0.000000325","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234","input_cache_read":"0.000000156","input_cache_write":"0.000000975"}]},"name":"Qwen: Qwen-Plus (cheapThink)","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","base_model":"qwen/qwen-plus","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"openai/o3-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"name":"OpenAI: o3 Mini (cheapThink)","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","base_model":"openai/o3-mini","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-small-24b-instruct-2501:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000008"},"name":"Mistral: Mistral Small 3 (cheapThink)","description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","base_model":"mistralai/mistral-small-24b-instruct-2501","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"perplexity/sonar:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":127072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"pricing":{"prompt":"0.000001","completion":"0.000001","web_search":"0.005"},"name":"Perplexity: Sonar (cheapThink)","description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","base_model":"perplexity/sonar","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":64000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000025"},"name":"DeepSeek: R1 (cheapThink)","description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","base_model":"deepseek/deepseek-r1","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-01:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000192,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["max_tokens","temperature","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000011"},"name":"MiniMax: MiniMax-01 (cheapThink)","description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","base_model":"minimax/minimax-01","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"microsoft/phi-4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":16384,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000014"},"name":"Microsoft: Phi 4 (cheapThink)","description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","base_model":"microsoft/phi-4","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-chat:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000002574","completion":"0.0000010287"},"name":"DeepSeek: DeepSeek V3 (cheapThink)","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","base_model":"deepseek/deepseek-chat","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"sao10k/l3.3-euryale-70b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000075"},"name":"Sao10K: Llama 3.3 Euryale 70B (cheapThink)","description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","base_model":"sao10k/l3.3-euryale-70b","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/o1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"pricing":{"prompt":"0.000015","completion":"0.00006","web_search":"0.01","input_cache_read":"0.0000075"},"name":"OpenAI: o1 (cheapThink)","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","base_model":"openai/o1","modifier":"cheapThink","request_multiplier":175,"required_plans":["ultra","enterprise"]},{"id":"cohere/command-r7b-12-2024:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000000375","completion":"0.00000015"},"name":"Cohere: Command R7B (12-2024) (cheapThink)","description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","base_model":"cohere/command-r7b-12-2024","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000032","input_cache_read":"0.00000011","discount":0},"name":"Meta: Llama 3.3 70B Instruct (cheapThink)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-lite-v1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":300000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000024"},"name":"Amazon: Nova Lite 1.0 (cheapThink)","description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","base_model":"amazon/nova-lite-v1","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-micro-v1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.000000035","completion":"0.00000014"},"name":"Amazon: Nova Micro 1.0 (cheapThink)","description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","base_model":"amazon/nova-micro-v1","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"amazon/nova-pro-v1:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":300000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"pricing":{"prompt":"0.0000008","completion":"0.0000032"},"name":"Amazon: Nova Pro 1.0 (cheapThink)","description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","base_model":"amazon/nova-pro-v1","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4o-2024-11-20:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o (2024-11-20) (cheapThink)","description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","base_model":"openai/gpt-4o-2024-11-20","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large 2407 (cheapThink)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-2.5-coder-32b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000066","completion":"0.000001"},"name":"Qwen2.5 Coder 32B Instruct (cheapThink)","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","base_model":"qwen/qwen-2.5-coder-32b-instruct","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"thedrummer/unslopnemo-12b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"name":"TheDrummer: UnslopNemo 12B (cheapThink)","description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","base_model":"thedrummer/unslopnemo-12b","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"anthracite-org/magnum-v4-72b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.000005"},"name":"Magnum v4 72B (cheapThink)","description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","base_model":"anthracite-org/magnum-v4-72b","modifier":"cheapThink","request_multiplier":21,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen-2.5-7b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"name":"Qwen: Qwen2.5 7B Instruct (cheapThink)","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","base_model":"qwen/qwen-2.5-7b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.2-1b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":60000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.000000027","completion":"0.000000201"},"name":"Meta: Llama 3.2 1B Instruct (cheapThink)","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","base_model":"meta-llama/llama-3.2-1b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.2-3b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000033"},"name":"Meta: Llama 3.2 3B Instruct (cheapThink)","description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","base_model":"meta-llama/llama-3.2-3b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen-2.5-72b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000036","completion":"0.0000004"},"name":"Qwen2.5 72B Instruct (cheapThink)","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","base_model":"qwen/qwen-2.5-72b-instruct","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"cohere/command-r-08-2024:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"name":"Cohere: Command R (08-2024) (cheapThink)","description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","base_model":"cohere/command-r-08-2024","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"cohere/command-r-plus-08-2024:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"name":"Cohere: Command R+ (08-2024) (cheapThink)","description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","base_model":"cohere/command-r-plus-08-2024","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"sao10k/l3.1-euryale-70b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000085","completion":"0.00000085"},"name":"Sao10K: Llama 3.1 Euryale 70B v2.2 (cheapThink)","description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","base_model":"sao10k/l3.1-euryale-70b","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-3-llama-3.1-70b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000007"},"name":"Nous: Hermes 3 70B Instruct (cheapThink)","description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","base_model":"nousresearch/hermes-3-llama-3.1-70b","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nousresearch/hermes-3-llama-3.1-405b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000001"},"name":"Nous: Hermes 3 405B Instruct (cheapThink)","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","base_model":"nousresearch/hermes-3-llama-3.1-405b","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"sao10k/l3-lunaris-8b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"name":"Sao10K: Llama 3 8B Lunaris (cheapThink)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-2024-08-06:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o (2024-08-06) (cheapThink)","description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","base_model":"openai/gpt-4o-2024-08-06","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"meta-llama/llama-3.1-70b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"name":"Meta: Llama 3.1 70B Instruct (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","base_model":"meta-llama/llama-3.1-70b-instruct","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000002","completion":"0.00000004","input_cache_read":"0.000000025","discount":0},"name":"Meta: Llama 3.1 8B Instruct (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000003","discount":0},"name":"Mistral: Mistral Nemo (cheapThink)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-mini:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini (cheapThink)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini-2024-07-18:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini (2024-07-18) (cheapThink)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini-2024-07-18","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"google/gemma-2-27b-it:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"pricing":{"prompt":"0.00000065","completion":"0.00000065"},"name":"Google: Gemma 2 27B (cheapThink)","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","base_model":"google/gemma-2-27b-it","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-4o:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"name":"OpenAI: GPT-4o (cheapThink)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","base_model":"openai/gpt-4o","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4o-2024-05-13:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.000005","completion":"0.000015"},"name":"OpenAI: GPT-4o (2024-05-13) (cheapThink)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","base_model":"openai/gpt-4o-2024-05-13","modifier":"cheapThink","request_multiplier":50,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral: Mixtral 8x22B Instruct (cheapThink)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"microsoft/wizardlm-2-8x22b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65535,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"pricing":{"prompt":"0.00000062","completion":"0.00000062"},"name":"WizardLM-2 8x22B (cheapThink)","description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","base_model":"microsoft/wizardlm-2-8x22b","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-4-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00001","completion":"0.00003"},"name":"OpenAI: GPT-4 Turbo (cheapThink)","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","base_model":"openai/gpt-4-turbo","modifier":"cheapThink","request_multiplier":100,"required_plans":["pro","ultra","enterprise"]},{"id":"mistralai/mistral-large:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large (cheapThink)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-0613:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":4095,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002"},"name":"OpenAI: GPT-3.5 Turbo (older v0613) (cheapThink)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","base_model":"openai/gpt-3.5-turbo-0613","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-instruct:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":4095,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.000002"},"name":"OpenAI: GPT-3.5 Turbo Instruct (cheapThink)","description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","base_model":"openai/gpt-3.5-turbo-instruct","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-3.5-turbo-16k:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":16385,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000004"},"name":"OpenAI: GPT-3.5 Turbo 16k (cheapThink)","description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","base_model":"openai/gpt-3.5-turbo-16k","modifier":"cheapThink","request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mancer/weaver:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000075"},"name":"Mancer: Weaver (alpha) (cheapThink)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","base_model":"mancer/weaver","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"undi95/remm-slerp-l2-13b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":6144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000035","completion":"0.00000065"},"name":"ReMM SLERP 13B (cheapThink)","description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","base_model":"undi95/remm-slerp-l2-13b","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"name":"MythoMax 13B (cheapThink)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-3.5-turbo:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":16385,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000015"},"name":"OpenAI: GPT-3.5 Turbo (cheapThink)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","base_model":"openai/gpt-3.5-turbo","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8191,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00003","completion":"0.00006"},"name":"OpenAI: GPT-4 (cheapThink)","description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","base_model":"openai/gpt-4","modifier":"cheapThink","request_multiplier":250,"required_plans":["enterprise"]},{"id":"anthropic/claude-sonnet-5.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5.5 (fast) (cheapThink)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5:fast","provider":"Amazon Bedrock","variant":"fast","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5.5 (cheapest-provider) (cheapThink)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5:cheapest-provider","provider":"Google","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5.5 (normal) (cheapThink)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","base_model":"anthropic/claude-sonnet-5.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008","discount":0},"name":"Anthropic: Claude Opus 5.5 (fast) (cheapThink)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008","discount":0},"name":"Anthropic: Claude Opus 5.5 (cheapest-provider) (cheapThink)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5:cheapest-provider","provider":"Amazon Bedrock","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"name":"Anthropic: Claude Opus 5.5 (normal) (cheapThink)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","base_model":"anthropic/claude-opus-5.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":53,"required_plans":["pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.6-flash:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (fast) (cheapThink)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (cheapest-provider) (cheapThink)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.000000003","discount":0},"name":"Xiaomi: MiMo-V2.6-Flash (quality) (cheapThink)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-flash-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"name":"Xiaomi: MiMo-V2.6-Flash (normal) (cheapThink)","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","base_model":"xiaomi/mimo-v2.6-flash-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (fast) (cheapThink)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000087","input_cache_read":"0.0000000036","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (cheapest-provider) (cheapThink)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.000000004","discount":0},"name":"Xiaomi: MiMo-V2.6-Pro (quality) (cheapThink)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.6-pro-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"name":"Xiaomi: MiMo-V2.6-Pro (normal) (cheapThink)","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","base_model":"xiaomi/mimo-v2.6-pro-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"x-ai/grok-4.7:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (fast) (cheapThink)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.7:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (cheapest-provider) (cheapThink)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.7-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.7 (normal) (cheapThink)","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","base_model":"x-ai/grok-4.7-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4.1-flash:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (fast) (cheapThink)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:fast","provider":"Decart","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (cheapest-provider) (cheapThink)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:cheapest-provider","provider":"Decart","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000098","completion":"0.00000039","input_cache_read":"0.000000009","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (quality) (cheapThink)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:quality","provider":"Morph","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000012563","completion":"0.00000132","input_cache_read":"0.000000012563"},"name":"DeepSeek: DeepSeek V4.1 Flash (normal) (cheapThink)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-fable-5.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5.1 (fast) (cheapThink)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1:fast","provider":"Google","variant":"fast","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5.1 (cheapest-provider) (cheapThink)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5.1 (normal) (cheapThink)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","base_model":"anthropic/claude-fable-5.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"tencent/hy4-preview:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (fast) (cheapThink)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:fast","provider":"SiliconFlow","variant":"fast","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (cheapest-provider) (cheapThink)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (quality) (cheapThink)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview (normal) (cheapThink)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview-normal","variant":"normal","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-flash:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003","discount":0},"name":"Z.ai: GLM 5.3 Flash (fast) (cheapThink)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash:fast","provider":"BaseTen","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3-flash:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015","discount":0.5},"name":"Z.ai: GLM 5.3 Flash (cheapest-provider) (cheapThink)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flash:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000105","completion":"0.00000035","input_cache_read":"0.0000000245","discount":0.3},"name":"Z.ai: GLM 5.3 Flash (quality) (cheapThink)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash:quality","provider":"Near AI","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-5.3-flash-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003"},"name":"Z.ai: GLM 5.3 Flash (normal) (cheapThink)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","base_model":"z-ai/glm-5.3-flash-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048575,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000014","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (fast) (cheapThink)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:fast","provider":"GMICloud","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686","discount":0.51},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (cheapest-provider) (cheapThink)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000028","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (quality) (cheapThink)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (normal) (cheapThink)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-5.3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000084","completion":"0.00000264","input_cache_read":"0.000000156","discount":0.4},"name":"Z.ai: GLM 5.3 (fast) (cheapThink)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3:fast","provider":"Phala","variant":"fast","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000042","completion":"0.00000132","input_cache_read":"0.000000078","discount":0.7},"name":"Z.ai: GLM 5.3 (cheapest-provider) (cheapThink)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3:cheapest-provider","provider":"Novita","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-5.3:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.3 (quality) (cheapThink)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.000007","input_cache_read":"0.000000065"},"name":"Z.ai: GLM 5.3 (normal) (cheapThink)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","base_model":"z-ai/glm-5.3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-27b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (fast) (cheapThink)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (cheapest-provider) (cheapThink)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:cheapest-provider","provider":"Phala","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (quality) (cheapThink)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:quality","provider":"DeepInfra","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000425","completion":"0.00000255","input_cache_read":"0.000000085","input_cache_write":"0.00000053125"},"name":"Qwen: Qwen3.8 27B (normal) (cheapThink)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (fast) (cheapThink)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b:fast","provider":"Novita","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (cheapest-provider) (cheapThink)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b:cheapest-provider","provider":"Novita","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","discount":0},"name":"Qwen: Qwen3.8 2.4T A95B (quality) (cheapThink)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.8-2.4t-a95b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"name":"Qwen: Qwen3.8 2.4T A95B (normal) (cheapThink)","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","base_model":"qwen/qwen3.8-2.4t-a95b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000957","completion":"0.0000028776","input_cache_read":"0.000000099","discount":0.34},"name":"DeepSeek: DeepSeek V4 Pro 0813 (fast) (cheapThink)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:fast","provider":"Phala","variant":"fast","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813 (cheapest-provider) (cheapThink)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:cheapest-provider","provider":"Ionstream","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001056","completion":"0.000003168","input_cache_read":"0.000000035","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0813 (quality) (cheapThink)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:quality","provider":"NextBit","variant":"quality","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}]},"name":"DeepSeek: DeepSeek V4 Pro 0813 (normal) (cheapThink)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813-normal","variant":"normal","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000022","completion":"0.0000066","web_search":"0.01","input_cache_read":"0.00000055","input_cache_write":"0","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000044","completion":"0.0000132","input_cache_read":"0.0000011","input_cache_write":"0"}]},"name":"SpaceXAI: Grok 4.6 (fast) (cheapThink)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6:fast","provider":"Amazon Bedrock","variant":"fast","modifier":"cheapThink","request_multiplier":22,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6 (cheapest-provider) (cheapThink)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.6-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"name":"SpaceXAI: Grok 4.6 (normal) (cheapThink)","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","base_model":"x-ai/grok-4.6-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3.5-lightning:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000016","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3.5 Lightning (fast) (cheapThink)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000049","completion":"0.00000014","input_cache_read":"0.0000000245","discount":0.3},"name":"NVIDIA: Nemotron 3.5 Lightning (cheapest-provider) (cheapThink)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning:cheapest-provider","provider":"Io Net","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.0000002","input_cache_read":"0.00000004","discount":0},"name":"NVIDIA: Nemotron 3.5 Lightning (quality) (cheapThink)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning:quality","provider":"CoreWeave","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3.5-lightning-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000016","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3.5 Lightning (normal) (cheapThink)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","base_model":"nvidia/nemotron-3.5-lightning-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta/muse-glimmer-30b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (fast) (cheapThink)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b:fast","provider":"Phala","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta/muse-glimmer-30b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000011","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (cheapest-provider) (cheapThink)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b:cheapest-provider","provider":"Phala","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta/muse-glimmer-30b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000004","discount":0},"name":"Meta: Muse Glimmer 30B (quality) (cheapThink)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b:quality","provider":"DeepInfra","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"meta/muse-glimmer-30b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000004"},"name":"Meta: Muse Glimmer 30B (normal) (cheapThink)","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","base_model":"meta/muse-glimmer-30b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.0000000238","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (fast) (cheapThink)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:fast","provider":"DigitalOcean","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731 (cheapest-provider) (cheapThink)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (quality) (cheapThink)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000118","completion":"0.00000128","input_cache_read":"0.0000000118"},"name":"DeepSeek: DeepSeek V4 Flash 0731 (normal) (cheapThink)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Opus 5 (fast) (cheapThink)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5:fast","provider":"Anthropic","variant":"fast","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-opus-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 5 (cheapest-provider) (cheapThink)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5 (normal) (cheapThink)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000201","completion":"0.00001005","input_cache_read":"0.000000201","discount":0.33},"name":"MoonshotAI: Kimi K3 (fast) (cheapThink)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:fast","provider":"Decart","variant":"fast","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3 (cheapest-provider) (cheapThink)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:cheapest-provider","provider":"Phala","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001274","completion":"0.000013296","input_cache_read":"0.000000278","discount":0},"name":"MoonshotAI: Kimi K3 (quality) (cheapThink)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:quality","provider":"Morph","variant":"quality","modifier":"cheapThink","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.000014","input_cache_read":"0.00000031"},"name":"MoonshotAI: Kimi K3 (normal) (cheapThink)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":28,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (fast) (cheapThink)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (cheapest-provider) (cheapThink)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":500000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"name":"SpaceXAI: Grok 4.5 (normal) (cheapThink)","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","base_model":"x-ai/grok-4.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000053","input_cache_read":"0.000000033","discount":0},"name":"Tencent: Hy3 (fast) (cheapThink)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","discount":0,"overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (cheapest-provider) (cheapThink)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:cheapest-provider","provider":"Tencent","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000058","input_cache_read":"0.000000035","discount":0},"name":"Tencent: Hy3 (quality) (cheapThink)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (normal) (cheapThink)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-sonnet-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (fast) (cheapThink)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5:fast","provider":"Amazon Bedrock","variant":"fast","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (cheapest-provider) (cheapThink)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5 (normal) (cheapThink)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000118","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.2 (fast) (cheapThink)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2:fast","provider":"Cloudflare","variant":"fast","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005625","completion":"0.0000018","input_cache_read":"0.000000105","discount":0.25},"name":"Z.ai: GLM 5.2 (cheapest-provider) (cheapThink)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000232","completion":"0.000003553","input_cache_read":"0.000000137","discount":0},"name":"Z.ai: GLM 5.2 (quality) (cheapThink)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2:quality","provider":"Morph","variant":"quality","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.2-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000152","completion":"0.000012","input_cache_read":"0.00000015"},"name":"Z.ai: GLM 5.2 (normal) (cheapThink)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","base_model":"z-ai/glm-5.2-normal","variant":"normal","modifier":"cheapThink","request_multiplier":21,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code (fast) (cheapThink)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code:fast","provider":"StreamLake","variant":"fast","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007125","completion":"0.000003","input_cache_read":"0.0000001425","discount":0.25},"name":"MoonshotAI: Kimi K2.7 Code (cheapest-provider) (cheapThink)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000085916","completion":"0.0000038","input_cache_read":"0.00000017993","discount":0},"name":"MoonshotAI: Kimi K2.7 Code (quality) (cheapThink)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.7-code-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006712","completion":"0.00000335","input_cache_read":"0.00000018"},"name":"MoonshotAI: Kimi K2.7 Code (normal) (cheapThink)","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","base_model":"moonshotai/kimi-k2.7-code-normal","variant":"normal","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-fable-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5 (fast) (cheapThink)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5:fast","provider":"Anthropic","variant":"fast","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Fable 5 (cheapest-provider) (cheapThink)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-fable-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"name":"Anthropic: Claude Fable 5 (normal) (cheapThink)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","base_model":"anthropic/claude-fable-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (fast) (cheapThink)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (cheapest-provider) (cheapThink)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000003125","input_cache_read":"0.0000001875","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (quality) (cheapThink)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:quality","provider":"Venice","variant":"quality","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra (normal) (cheapThink)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000028","completion":"0.0000011","input_cache_read":"0.000000056","discount":0},"name":"MiniMax: MiniMax M3 (fast) (cheapThink)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.00000096","input_cache_read":"0.00000005","discount":0},"name":"MiniMax: MiniMax M3 (cheapest-provider) (cheapThink)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3:cheapest-provider","provider":"CoreWeave","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m3:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M3 (quality) (cheapThink)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006"},"name":"MiniMax: MiniMax M3 (normal) (cheapThink)","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","base_model":"minimax/minimax-m3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-opus-4.8:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (fast) (cheapThink)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (cheapest-provider) (cheapThink)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8 (normal) (cheapThink)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8-normal","variant":"normal","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (fast) (cheapThink)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (cheapest-provider) (cheapThink)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-build-0.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok Build 0.1 (normal) (cheapThink)","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","base_model":"x-ai/grok-build-0.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (fast) (cheapThink)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (cheapest-provider) (cheapThink)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (normal) (cheapThink)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075","discount":0},"name":"Mistral: Mistral Medium 3.5 (fast) (cheapThink)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075","discount":0},"name":"Mistral: Mistral Medium 3.5 (cheapest-provider) (cheapThink)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"name":"Mistral: Mistral Medium 3.5 (normal) (cheapThink)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","base_model":"mistralai/mistral-medium-3-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-35b-a3b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.000001","discount":0},"name":"Qwen: Qwen3.6 35B A3B (fast) (cheapThink)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b:fast","provider":"Venice","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000007","input_cache_read":"0.000000025","discount":0},"name":"Qwen: Qwen3.6 35B A3B (cheapest-provider) (cheapThink)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.6 35B A3B (quality) (cheapThink)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-35b-a3b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005"},"name":"Qwen: Qwen3.6 35B A3B (normal) (cheapThink)","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","base_model":"qwen/qwen3.6-35b-a3b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.6-27b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.0000027","discount":0},"name":"Qwen: Qwen3.6 27B (fast) (cheapThink)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b:fast","provider":"Alibaba","variant":"fast","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000002","input_cache_read":"0.00000003","discount":0},"name":"Qwen: Qwen3.6 27B (cheapest-provider) (cheapThink)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b:cheapest-provider","provider":"Chutes","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000032","discount":0},"name":"Qwen: Qwen3.6 27B (quality) (cheapThink)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.6-27b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000032","completion":"0.00000325","input_cache_read":"0.00000003"},"name":"Qwen: Qwen3.6 27B (normal) (cheapThink)","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","base_model":"qwen/qwen3.6-27b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000150162","completion":"0.000003135","input_cache_read":"0.000000135","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0423 (fast) (cheapThink)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro:fast","provider":"SiliconFlow","variant":"fast","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174","discount":0.88},"name":"DeepSeek: DeepSeek V4 Pro 0423 (cheapest-provider) (cheapThink)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-pro:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000348","input_cache_read":"0.0000001","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0423 (quality) (cheapThink)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"name":"DeepSeek: DeepSeek V4 Pro 0423 (normal) (cheapThink)","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","base_model":"deepseek/deepseek-v4-pro-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048575,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000091","completion":"0.000000182","input_cache_read":"0.0000000182","discount":0.35},"name":"DeepSeek: DeepSeek V4 Flash 0423 (fast) (cheapThink)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash:fast","provider":"GMICloud","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.000000084","input_cache_read":"0.0000000084","discount":0.7},"name":"DeepSeek: DeepSeek V4 Flash 0423 (cheapest-provider) (cheapThink)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.00000007","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0423 (quality) (cheapThink)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000106","completion":"0.00000128","input_cache_read":"0.0000000106"},"name":"DeepSeek: DeepSeek V4 Flash 0423 (normal) (cheapThink)","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","base_model":"deepseek/deepseek-v4-flash-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000048","completion":"0.0000018","input_cache_read":"0.000000096","discount":0},"name":"Xiaomi: MiMo-V2.5-Pro (fast) (cheapThink)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro:fast","provider":"DigitalOcean","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.5-pro:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro (cheapest-provider) (cheapThink)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro:cheapest-provider","provider":"GMICloud","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003045","completion":"0.000000609","input_cache_read":"0.0000000028","discount":0.3},"name":"Xiaomi: MiMo-V2.5-Pro (quality) (cheapThink)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"xiaomi/mimo-v2.5-pro-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"name":"Xiaomi: MiMo-V2.5-Pro (normal) (cheapThink)","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","base_model":"xiaomi/mimo-v2.5-pro-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"xiaomi/mimo-v2.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000008","discount":0},"name":"Xiaomi: MiMo-V2.5 (fast) (cheapThink)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5:fast","provider":"Venice","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"xiaomi/mimo-v2.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5 (cheapest-provider) (cheapThink)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5:cheapest-provider","provider":"GMICloud","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.5:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.00000000255","discount":0.15},"name":"Xiaomi: MiMo-V2.5 (quality) (cheapThink)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"xiaomi/mimo-v2.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"name":"Xiaomi: MiMo-V2.5 (normal) (cheapThink)","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","base_model":"xiaomi/mimo-v2.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"moonshotai/kimi-k2.6:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000057","completion":"0.0000024","input_cache_read":"0.000000114","discount":0},"name":"MoonshotAI: Kimi K2.6 (fast) (cheapThink)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6:fast","provider":"DigitalOcean","variant":"fast","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000465","completion":"0.00000245","input_cache_read":"0.0000000975","discount":0},"name":"MoonshotAI: Kimi K2.6 (cheapest-provider) (cheapThink)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6:cheapest-provider","provider":"Inceptron","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.0000035","input_cache_read":"0.00000035","discount":0},"name":"MoonshotAI: Kimi K2.6 (quality) (cheapThink)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6:quality","provider":"Crusoe","variant":"quality","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.6-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.000004","input_cache_read":"0.00000016"},"name":"MoonshotAI: Kimi K2.6 (normal) (cheapThink)","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","base_model":"moonshotai/kimi-k2.6-normal","variant":"normal","modifier":"cheapThink","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (fast) (cheapThink)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (cheapest-provider) (cheapThink)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7 (normal) (cheapThink)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7-normal","variant":"normal","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026","discount":0},"name":"Z.ai: GLM 5.1 (fast) (cheapThink)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1:fast","provider":"Friendli","variant":"fast","modifier":"cheapThink","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000966","completion":"0.000003036","input_cache_read":"0.0000001794","discount":0.31},"name":"Z.ai: GLM 5.1 (cheapest-provider) (cheapThink)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000119","completion":"0.00000374","input_cache_read":"0.0000006","discount":0},"name":"Z.ai: GLM 5.1 (quality) (cheapThink)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000026"},"name":"Z.ai: GLM 5.1 (normal) (cheapThink)","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","base_model":"z-ai/glm-5.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":14,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"google/gemma-4-26b-a4b-it:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 26B A4B  (fast) (cheapThink)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it:fast","provider":"CoreWeave","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000042","completion":"0.00000022","input_cache_read":"0.000000021","discount":0},"name":"Google: Gemma 4 26B A4B  (cheapest-provider) (cheapThink)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 26B A4B  (quality) (cheapThink)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it:quality","provider":"CoreWeave","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-26b-a4b-it-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 26B A4B  (normal) (cheapThink)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","base_model":"google/gemma-4-26b-a4b-it-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000004","discount":0},"name":"Google: Gemma 4 31B (fast) (cheapThink)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it:fast","provider":"Friendli","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005","discount":0},"name":"Google: Gemma 4 31B (cheapest-provider) (cheapThink)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000004","input_cache_read":"0.00000014","discount":0},"name":"Google: Gemma 4 31B (quality) (cheapThink)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it:quality","provider":"Crusoe","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-4-31b-it-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"name":"Google: Gemma 4 31B (normal) (cheapThink)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","base_model":"google/gemma-4-31b-it-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"x-ai/grok-4.20-multi-agent:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (fast) (cheapThink)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-multi-agent:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (cheapest-provider) (cheapThink)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-multi-agent-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 Multi-Agent (normal) (cheapThink)","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","base_model":"x-ai/grok-4.20-multi-agent-normal","variant":"normal","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (fast) (cheapThink)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20:fast","provider":"xAI","variant":"fast","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (cheapest-provider) (cheapThink)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.20-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":2000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.20 (normal) (cheapThink)","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","base_model":"x-ai/grok-4.20-normal","variant":"normal","modifier":"cheapThink","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.7:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042","discount":0.3},"name":"MiniMax: MiniMax M2.7 (fast) (cheapThink)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7:fast","provider":"GMICloud","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2.7:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042","discount":0.3},"name":"MiniMax: MiniMax M2.7 (cheapest-provider) (cheapThink)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7:cheapest-provider","provider":"GMICloud","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"minimax/minimax-m2.7:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M2.7 (quality) (cheapThink)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7:quality","provider":"Minimax","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.7-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042"},"name":"MiniMax: MiniMax M2.7 (normal) (cheapThink)","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","base_model":"minimax/minimax-m2.7-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Small 4 (fast) (cheapThink)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Small 4 (cheapest-provider) (cheapThink)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-small-2603-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Mistral: Mistral Small 4 (normal) (cheapThink)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","base_model":"mistralai/mistral-small-2603-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-9b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B (fast) (cheapThink)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b:fast","provider":"Darkbloom","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000013","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-9B (cheapest-provider) (cheapThink)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000025","discount":0},"name":"Qwen: Qwen3.5-9B (quality) (cheapThink)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-9b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000015"},"name":"Qwen: Qwen3.5-9B (normal) (cheapThink)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","base_model":"qwen/qwen3.5-9b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (fast) (cheapThink)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (cheapest-provider) (cheapThink)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3.5-35B-A3B (quality) (cheapThink)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-35b-a3b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004"},"name":"Qwen: Qwen3.5-35B-A3B (normal) (cheapThink)","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","base_model":"qwen/qwen3.5-35b-a3b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.5-27b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.0000026","discount":0},"name":"Qwen: Qwen3.5-27B (fast) (cheapThink)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-27b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156","discount":0},"name":"Qwen: Qwen3.5-27B (cheapest-provider) (cheapThink)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b:cheapest-provider","provider":"Alibaba","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-27b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000024","discount":0},"name":"Qwen: Qwen3.5-27B (quality) (cheapThink)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-27b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"name":"Qwen: Qwen3.5-27B (normal) (cheapThink)","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","base_model":"qwen/qwen3.5-27b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.5-122b-a10b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (fast) (cheapThink)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b:fast","provider":"SiliconFlow","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (cheapest-provider) (cheapThink)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b:cheapest-provider","provider":"SiliconFlow","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000032","discount":0},"name":"Qwen: Qwen3.5-122B-A10B (quality) (cheapThink)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-122b-a10b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"name":"Qwen: Qwen3.5-122B-A10B (normal) (cheapThink)","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","base_model":"qwen/qwen3.5-122b-a10b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (fast) (cheapThink)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (cheapest-provider) (cheapThink)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex (normal) (cheapThink)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex-normal","variant":"normal","modifier":"cheapThink","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (fast) (cheapThink)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (cheapest-provider) (cheapThink)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6 (normal) (cheapThink)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6-normal","variant":"normal","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000036","input_cache_read":"0.0000003","discount":0},"name":"Qwen: Qwen3.5 397B A17B (fast) (cheapThink)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000039","completion":"0.00000234","discount":0},"name":"Qwen: Qwen3.5 397B A17B (cheapest-provider) (cheapThink)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b:cheapest-provider","provider":"Alibaba","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000036","input_cache_read":"0.0000003","discount":0},"name":"Qwen: Qwen3.5 397B A17B (quality) (cheapThink)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3.5-397b-a17b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.000003","input_cache_read":"0.00000022"},"name":"Qwen: Qwen3.5 397B A17B (normal) (cheapThink)","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","base_model":"qwen/qwen3.5-397b-a17b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"minimax/minimax-m2.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":196608,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000295","completion":"0.0000012","input_cache_read":"0.00000006","discount":0},"name":"MiniMax: MiniMax M2.5 (fast) (cheapThink)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5:fast","provider":"AtlasCloud","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000095","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5 (cheapest-provider) (cheapThink)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5:cheapest-provider","provider":"Venice","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2.5:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.5 (quality) (cheapThink)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000108","input_cache_read":"0.000000027"},"name":"MiniMax: MiniMax M2.5 (normal) (cheapThink)","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","base_model":"minimax/minimax-m2.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"z-ai/glm-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000007","completion":"0.00000224","input_cache_read":"0.00000014","discount":0},"name":"Z.ai: GLM 5 (fast) (cheapThink)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5:fast","provider":"Baidu","variant":"fast","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012","discount":0.4},"name":"Z.ai: GLM 5 (cheapest-provider) (cheapThink)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.00000255","input_cache_read":"0.0000002","discount":0},"name":"Z.ai: GLM 5 (quality) (cheapThink)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":9,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"name":"Z.ai: GLM 5 (normal) (cheapThink)","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","base_model":"z-ai/glm-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (fast) (cheapThink)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6:fast","provider":"Google","variant":"fast","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (cheapest-provider) (cheapThink)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6 (normal) (cheapThink)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6-normal","variant":"normal","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder-next:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 Coder Next (fast) (cheapThink)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next:fast","provider":"Novita","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder-next:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Coder Next (cheapest-provider) (cheapThink)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next:cheapest-provider","provider":"Parasail","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-coder-next:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Coder Next (quality) (cheapThink)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-coder-next-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Coder Next (normal) (cheapThink)","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","base_model":"qwen/qwen3-coder-next-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"moonshotai/kimi-k2.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (fast) (cheapThink)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5:fast","provider":"SiliconFlow","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (cheapest-provider) (cheapThink)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5:cheapest-provider","provider":"SiliconFlow","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007","discount":0},"name":"MoonshotAI: Kimi K2.5 (quality) (cheapThink)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k2.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007"},"name":"MoonshotAI: Kimi K2.5 (normal) (cheapThink)","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","base_model":"moonshotai/kimi-k2.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-flash:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000605","completion":"0.0000004","discount":0},"name":"Z.ai: GLM 4.7 Flash (fast) (cheapThink)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash:fast","provider":"Cloudflare","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash (cheapest-provider) (cheapThink)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash:cheapest-provider","provider":"Venice","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.0000004","input_cache_read":"0.00000001","discount":0},"name":"Z.ai: GLM 4.7 Flash (quality) (cheapThink)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.7-flash-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000605","completion":"0.0000004"},"name":"Z.ai: GLM 4.7 Flash (normal) (cheapThink)","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","base_model":"z-ai/glm-4.7-flash-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (fast) (cheapThink)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1:fast","provider":"Novita","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (cheapest-provider) (cheapThink)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1:cheapest-provider","provider":"Novita","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003","discount":0},"name":"MiniMax: MiniMax M2.1 (quality) (cheapThink)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"minimax/minimax-m2.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"name":"MiniMax: MiniMax M2.1 (normal) (cheapThink)","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","base_model":"minimax/minimax-m2.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"z-ai/glm-4.7:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","discount":0},"name":"Z.ai: GLM 4.7 (fast) (cheapThink)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7:fast","provider":"Google","variant":"fast","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.7 (cheapest-provider) (cheapThink)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.00000054","completion":"0.00000198","input_cache_read":"0.000000099","discount":0.1},"name":"Z.ai: GLM 4.7 (quality) (cheapThink)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.7-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"name":"Z.ai: GLM 4.7 (normal) (cheapThink)","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","base_model":"z-ai/glm-4.7-normal","variant":"normal","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-nano-30b-a3b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (fast) (cheapThink)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b:fast","provider":"Crusoe","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (cheapest-provider) (cheapThink)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b:cheapest-provider","provider":"Crusoe","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003","discount":0},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (quality) (cheapThink)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b:quality","provider":"Crusoe","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"nvidia/nemotron-3-nano-30b-a3b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (normal) (cheapThink)","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","base_model":"nvidia/nemotron-3-nano-30b-a3b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Ministral 3 14B 2512 (fast) (cheapThink)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Ministral 3 14B 2512 (cheapest-provider) (cheapThink)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-14b-2512-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"name":"Mistral: Ministral 3 14B 2512 (normal) (cheapThink)","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","base_model":"mistralai/ministral-14b-2512-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Ministral 3 8B 2512 (fast) (cheapThink)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Ministral 3 8B 2512 (cheapest-provider) (cheapThink)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-8b-2512-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"name":"Mistral: Ministral 3 8B 2512 (normal) (cheapThink)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-8b-2512-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Ministral 3 3B 2512 (fast) (cheapThink)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Ministral 3 3B 2512 (cheapest-provider) (cheapThink)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/ministral-3b-2512-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"name":"Mistral: Ministral 3 3B 2512 (normal) (cheapThink)","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","base_model":"mistralai/ministral-3b-2512-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v3.2:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000056","completion":"0.00000168","discount":0},"name":"DeepSeek: DeepSeek V3.2 (fast) (cheapThink)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2:fast","provider":"Google","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v3.2:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002088","completion":"0.0000003096","input_cache_read":"0.0000000216","discount":0.28},"name":"DeepSeek: DeepSeek V3.2 (cheapest-provider) (cheapThink)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2:cheapest-provider","provider":"GMICloud","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v3.2:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000259","completion":"0.00000042","input_cache_read":"0.000000135","discount":0},"name":"DeepSeek: DeepSeek V3.2 (quality) (cheapThink)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v3.2-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000028","completion":"0.00000042","input_cache_read":"0.000000028"},"name":"DeepSeek: DeepSeek V3.2 (normal) (cheapThink)","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","base_model":"deepseek/deepseek-v3.2-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-4.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.5 (fast) (cheapThink)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5:fast","provider":"Amazon Bedrock","variant":"fast","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.5 (cheapest-provider) (cheapThink)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.5 (normal) (cheapThink)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","base_model":"anthropic/claude-opus-4.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"mistralai/voxtral-small-24b-2507:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Voxtral Small 24B 2507 (fast) (cheapThink)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/voxtral-small-24b-2507:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001","discount":0},"name":"Mistral: Voxtral Small 24B 2507 (cheapest-provider) (cheapThink)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/voxtral-small-24b-2507-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","audio","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"name":"Mistral: Voxtral Small 24B 2507 (normal) (cheapThink)","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","base_model":"mistralai/voxtral-small-24b-2507-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"minimax/minimax-m2:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (fast) (cheapThink)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2:fast","provider":"Minimax","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (cheapest-provider) (cheapThink)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2:cheapest-provider","provider":"Minimax","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000255","completion":"0.00000102","discount":0.15},"name":"MiniMax: MiniMax M2 (quality) (cheapThink)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2:quality","provider":"Minimax","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"minimax/minimax-m2-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000012"},"name":"MiniMax: MiniMax M2 (normal) (cheapThink)","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","base_model":"minimax/minimax-m2-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-haiku-4.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002","discount":0},"name":"Anthropic: Claude Haiku 4.5 (fast) (cheapThink)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002","discount":0},"name":"Anthropic: Claude Haiku 4.5 (cheapest-provider) (cheapThink)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-haiku-4.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":200000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"name":"Anthropic: Claude Haiku 4.5 (normal) (cheapThink)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","base_model":"anthropic/claude-haiku-4.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-30b-a3b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (fast) (cheapThink)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct:fast","provider":"Alibaba","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (cheapest-provider) (cheapThink)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct:cheapest-provider","provider":"Alibaba","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000007","discount":0},"name":"Qwen: Qwen3 VL 30B A3B Instruct (quality) (cheapThink)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-30b-a3b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"name":"Qwen: Qwen3 VL 30B A3B Instruct (normal) (cheapThink)","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","base_model":"qwen/qwen3-vl-30b-a3b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.6:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":202752,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000002","input_cache_read":"0.0000001","discount":0},"name":"Z.ai: GLM 4.6 (fast) (cheapThink)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":198000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008","discount":0},"name":"Z.ai: GLM 4.6 (cheapest-provider) (cheapThink)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6:cheapest-provider","provider":"Venice","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000055","completion":"0.0000022","input_cache_read":"0.00000011","discount":0},"name":"Z.ai: GLM 4.6 (quality) (cheapThink)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"z-ai/glm-4.6-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":204800,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008"},"name":"Z.ai: GLM 4.6 (normal) (cheapThink)","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","base_model":"z-ai/glm-4.6-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (fast) (cheapThink)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (cheapest-provider) (cheapThink)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"name":"Anthropic: Claude Sonnet 4.5 (normal) (cheapThink)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","base_model":"anthropic/claude-sonnet-4.5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-vl-235b-a22b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000026","completion":"0.00000104","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (fast) (cheapThink)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct:fast","provider":"Alibaba","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.00000088","input_cache_read":"0.00000011","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (cheapest-provider) (cheapThink)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 VL 235B A22B Instruct (quality) (cheapThink)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-vl-235b-a22b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000021","completion":"0.0000019","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 VL 235B A22B Instruct (normal) (cheapThink)","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","base_model":"qwen/qwen3-vl-235b-a22b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (fast) (cheapThink)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000975","completion":"0.00000078","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (cheapest-provider) (cheapThink)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct:cheapest-provider","provider":"Alibaba","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015","discount":0},"name":"Qwen: Qwen3 Next 80B A3B Instruct (quality) (cheapThink)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-next-80b-a3b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007"},"name":"Qwen: Qwen3 Next 80B A3B Instruct (normal) (cheapThink)","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","base_model":"qwen/qwen3-next-80b-a3b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013","discount":0},"name":"DeepSeek: DeepSeek V3.1 (fast) (cheapThink)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013","discount":0},"name":"DeepSeek: DeepSeek V3.1 (cheapest-provider) (cheapThink)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.000001","discount":0},"name":"DeepSeek: DeepSeek V3.1 (quality) (cheapThink)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"deepseek/deepseek-chat-v3.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"name":"DeepSeek: DeepSeek V3.1 (normal) (cheapThink)","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","base_model":"deepseek/deepseek-chat-v3.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"mistralai/mistral-medium-3.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.0000022","input_cache_read":"0.000000044","discount":0},"name":"Mistral: Mistral Medium 3.1 (fast) (cheapThink)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3.1 (cheapest-provider) (cheapThink)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3.1 (normal) (cheapThink)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","base_model":"mistralai/mistral-medium-3.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125","discount":0},"name":"OpenAI: GPT-5 (fast) (cheapThink)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5:fast","provider":"OpenAI","variant":"fast","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125","discount":0},"name":"OpenAI: GPT-5 (cheapest-provider) (cheapThink)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"name":"OpenAI: GPT-5 (normal) (cheapThink)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","base_model":"openai/gpt-5-normal","variant":"normal","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-oss-120b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","input_cache_read":"0.00000005","discount":0},"name":"OpenAI: gpt-oss-120b (fast) (cheapThink)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b:fast","provider":"Crusoe","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000017","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: gpt-oss-120b (cheapest-provider) (cheapThink)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b:cheapest-provider","provider":"CoreWeave","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","input_cache_read":"0.00000005","discount":0},"name":"OpenAI: gpt-oss-120b (quality) (cheapThink)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b:quality","provider":"Crusoe","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-120b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000037","completion":"0.00000017"},"name":"OpenAI: gpt-oss-120b (normal) (cheapThink)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","base_model":"openai/gpt-oss-120b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000014","discount":0},"name":"OpenAI: gpt-oss-20b (fast) (cheapThink)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009","discount":0},"name":"OpenAI: gpt-oss-20b (cheapest-provider) (cheapThink)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b:cheapest-provider","provider":"Darkbloom","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000003","completion":"0.00000014","discount":0},"name":"OpenAI: gpt-oss-20b (quality) (cheapThink)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b:quality","provider":"DeepInfra","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-oss-20b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"name":"OpenAI: gpt-oss-20b (normal) (cheapThink)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","base_model":"openai/gpt-oss-20b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (fast) (cheapThink)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct:fast","provider":"SiliconFlow","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":160000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000027","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (cheapest-provider) (cheapThink)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct:cheapest-provider","provider":"Novita","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (quality) (cheapThink)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-coder-30b-a3b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000007","completion":"0.00000028"},"name":"Qwen: Qwen3 Coder 30B A3B Instruct (normal) (cheapThink)","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","base_model":"qwen/qwen3-coder-30b-a3b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000052","discount":0},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (fast) (cheapThink)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507:fast","provider":"Alibaba","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004815","completion":"0.00000019305","discount":0.55},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (cheapest-provider) (cheapThink)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","discount":0},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (quality) (cheapThink)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-30b-a3b-instruct-2507-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Qwen: Qwen3 30B A3B Instruct 2507 (normal) (cheapThink)","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","base_model":"qwen/qwen3-30b-a3b-instruct-2507-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"z-ai/glm-4.5-air:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (fast) (cheapThink)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air:fast","provider":"Novita","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (cheapest-provider) (cheapThink)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air:cheapest-provider","provider":"Novita","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025","discount":0},"name":"Z.ai: GLM 4.5 Air (quality) (cheapThink)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"z-ai/glm-4.5-air-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"name":"Z.ai: GLM 4.5 Air (normal) (cheapThink)","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","base_model":"z-ai/glm-4.5-air-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-thinking-2507:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (fast) (cheapThink)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507:fast","provider":"Alibaba","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (cheapest-provider) (cheapThink)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507:cheapest-provider","provider":"Alibaba","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000003","discount":0},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (quality) (cheapThink)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-235b-a22b-thinking-2507-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000023","completion":"0.0000023"},"name":"Qwen: Qwen3 235B A22B Thinking 2507 (normal) (cheapThink)","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","base_model":"qwen/qwen3-235b-a22b-thinking-2507-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-coder:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000018","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (fast) (cheapThink)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder:fast","provider":"Google","variant":"fast","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (cheapest-provider) (cheapThink)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-coder:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000038","completion":"0.00000155","discount":0},"name":"Qwen: Qwen3 Coder 480B A35B (quality) (cheapThink)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3-coder-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"name":"Qwen: Qwen3 Coder 480B A35B (normal) (cheapThink)","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","base_model":"qwen/qwen3-coder-normal","variant":"normal","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.0000008","input_cache_read":"0.00000005","discount":0},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (fast) (cheapThink)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (cheapest-provider) (cheapThink)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507:cheapest-provider","provider":"GMICloud","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175","discount":0.75},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (quality) (cheapThink)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507:quality","provider":"GMICloud","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-235b-a22b-2507-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000055"},"name":"Qwen: Qwen3 235B A22B Instruct 2507 (normal) (cheapThink)","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","base_model":"qwen/qwen3-235b-a22b-2507-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Mistral: Mistral Small 3.2 24B (fast) (cheapThink)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000002","discount":0},"name":"Mistral: Mistral Small 3.2 24B (cheapest-provider) (cheapThink)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.0000003","input_cache_read":"0.00000005","discount":0},"name":"Mistral: Mistral Small 3.2 24B (quality) (cheapThink)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct:quality","provider":"Parasail","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-small-3.2-24b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009375","completion":"0.00000025"},"name":"Mistral: Mistral Small 3.2 24B (normal) (cheapThink)","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","base_model":"mistralai/mistral-small-3.2-24b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-r1-0528:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035","discount":0},"name":"DeepSeek: R1 0528 (fast) (cheapThink)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035","discount":0},"name":"DeepSeek: R1 0528 (cheapest-provider) (cheapThink)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000218","discount":0},"name":"DeepSeek: R1 0528 (quality) (cheapThink)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-r1-0528-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":163840,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"supported_efforts":[]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"name":"DeepSeek: R1 0528 (normal) (cheapThink)","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","base_model":"deepseek/deepseek-r1-0528-normal","variant":"normal","modifier":"cheapThink","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3 (fast) (cheapThink)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004","discount":0},"name":"Mistral: Mistral Medium 3 (cheapest-provider) (cheapThink)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-medium-3-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"name":"Mistral: Mistral Medium 3 (normal) (cheapThink)","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","base_model":"mistralai/mistral-medium-3-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"qwen/qwen3-14b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B (fast) (cheapThink)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b:fast","provider":"NextBit","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000022","discount":0},"name":"Qwen: Qwen3 14B (cheapest-provider) (cheapThink)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b:cheapest-provider","provider":"NextBit","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.00000024","discount":0},"name":"Qwen: Qwen3 14B (quality) (cheapThink)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b:quality","provider":"DeepInfra","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-14b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000012","completion":"0.00000024"},"name":"Qwen: Qwen3 14B (normal) (cheapThink)","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-14b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 32B (fast) (cheapThink)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":40960,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028","discount":0},"name":"Qwen: Qwen3 32B (cheapest-provider) (cheapThink)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"qwen/qwen3-32b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000057","discount":0},"name":"Qwen: Qwen3 32B (quality) (cheapThink)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b:quality","provider":"SiliconFlow","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3-32b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"name":"Qwen: Qwen3 32B (normal) (cheapThink)","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","base_model":"qwen/qwen3-32b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000022","completion":"0.0000088","web_search":"0.01","input_cache_read":"0.00000055","discount":0},"name":"OpenAI: GPT-4.1 (fast) (cheapThink)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005","discount":0},"name":"OpenAI: GPT-4.1 (cheapest-provider) (cheapThink)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"name":"OpenAI: GPT-4.1 (normal) (cheapThink)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","base_model":"openai/gpt-4.1-normal","variant":"normal","modifier":"cheapThink","request_multiplier":23,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001","discount":0},"name":"OpenAI: GPT-4.1 Mini (fast) (cheapThink)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001","discount":0},"name":"OpenAI: GPT-4.1 Mini (cheapest-provider) (cheapThink)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-mini-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"name":"OpenAI: GPT-4.1 Mini (normal) (cheapThink)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","base_model":"openai/gpt-4.1-mini-normal","variant":"normal","modifier":"cheapThink","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-4.1-nano:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025","discount":0},"name":"OpenAI: GPT-4.1 Nano (fast) (cheapThink)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano:fast","provider":"OpenAI","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1-nano:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.00000003","discount":0},"name":"OpenAI: GPT-4.1 Nano (cheapest-provider) (cheapThink)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4.1-nano-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1047576,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"name":"OpenAI: GPT-4.1 Nano (normal) (cheapThink)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","base_model":"openai/gpt-4.1-nano-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-maverick:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000035","completion":"0.000001","input_cache_read":"0.00000017","discount":0},"name":"Meta: Llama 4 Maverick (fast) (cheapThink)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-4-maverick:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525","discount":0},"name":"Meta: Llama 4 Maverick (cheapest-provider) (cheapThink)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick:cheapest-provider","provider":"DigitalOcean","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-maverick:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000027","completion":"0.00000085","discount":0},"name":"Meta: Llama 4 Maverick (quality) (cheapThink)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"meta-llama/llama-4-maverick-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001875","completion":"0.0000006525"},"name":"Meta: Llama 4 Maverick (normal) (cheapThink)","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","base_model":"meta-llama/llama-4-maverick-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":327680,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Meta: Llama 4 Scout (fast) (cheapThink)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-scout:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":327680,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Meta: Llama 4 Scout (cheapest-provider) (cheapThink)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-4-scout:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.00000059","discount":0},"name":"Meta: Llama 4 Scout (quality) (cheapThink)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-4-scout-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":1310720,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"name":"Meta: Llama 4 Scout (normal) (cheapThink)","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","base_model":"meta-llama/llama-4-scout-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":110000,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000003","discount":0},"name":"Google: Gemma 3 27B (fast) (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it:fast","provider":"Nebius","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000016","discount":0},"name":"Google: Gemma 3 27B (cheapest-provider) (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":98304,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.0000002","discount":0},"name":"Google: Gemma 3 27B (quality) (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"google/gemma-3-27b-it-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000045","input_cache_read":"0.00000004"},"name":"Google: Gemma 3 27B (normal) (cheapThink)","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","base_model":"google/gemma-3-27b-it-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-saba:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Saba (fast) (cheapThink)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-saba:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002","discount":0},"name":"Mistral: Saba (cheapest-provider) (cheapThink)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-saba-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32768,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"name":"Mistral: Saba (normal) (cheapThink)","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","base_model":"mistralai/mistral-saba-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000005","input_cache_read":"0.00000011","discount":0},"name":"Meta: Llama 3.3 70B Instruct (fast) (cheapThink)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct:fast","provider":"Parasail","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.00000032","discount":0},"name":"Meta: Llama 3.3 70B Instruct (cheapest-provider) (cheapThink)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":12288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000135","completion":"0.0000004","discount":0},"name":"Meta: Llama 3.3 70B Instruct (quality) (cheapThink)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.3-70b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.0000005","input_cache_read":"0.00000011"},"name":"Meta: Llama 3.3 70B Instruct (normal) (cheapThink)","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","base_model":"meta-llama/llama-3.3-70b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mistral-large-2407:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral Large 2407 (fast) (cheapThink)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral Large 2407 (cheapest-provider) (cheapThink)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mistral-large-2407-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral Large 2407 (normal) (cheapThink)","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","base_model":"mistralai/mistral-large-2407-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"sao10k/l3-lunaris-8b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (fast) (cheapThink)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b:fast","provider":"Novita","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (cheapest-provider) (cheapThink)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b:cheapest-provider","provider":"Parasail","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000005","discount":0},"name":"Sao10K: Llama 3 8B Lunaris (quality) (cheapThink)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b:quality","provider":"Novita","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"sao10k/l3-lunaris-8b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"name":"Sao10K: Llama 3 8B Lunaris (normal) (cheapThink)","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","base_model":"sao10k/l3-lunaris-8b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":32000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000152","completion":"0.000000287","discount":0},"name":"Meta: Llama 3.1 8B Instruct (fast) (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct:fast","provider":"Cloudflare","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000002","completion":"0.00000004","discount":0},"name":"Meta: Llama 3.1 8B Instruct (cheapest-provider) (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000022","completion":"0.00000022","input_cache_read":"0.00000022","discount":0},"name":"Meta: Llama 3.1 8B Instruct (quality) (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct:quality","provider":"CoreWeave","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"meta-llama/llama-3.1-8b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.00000008","input_cache_read":"0.000000025"},"name":"Meta: Llama 3.1 8B Instruct (normal) (cheapThink)","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","base_model":"meta-llama/llama-3.1-8b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000023","completion":"0.00000003","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Nemo (fast) (cheapThink)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo:fast","provider":"Io Net","variant":"fast","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000018","completion":"0.00000003","discount":0},"name":"Mistral: Mistral Nemo (cheapest-provider) (cheapThink)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo:cheapest-provider","provider":"DekaLLM","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000023","completion":"0.00000003","input_cache_read":"0.000000015","discount":0},"name":"Mistral: Mistral Nemo (quality) (cheapThink)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo:quality","provider":"Io Net","variant":"quality","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"mistralai/mistral-nemo-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000019","completion":"0.00000003"},"name":"Mistral: Mistral Nemo (normal) (cheapThink)","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","base_model":"mistralai/mistral-nemo-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-4o-mini:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.000000165","completion":"0.00000066","input_cache_read":"0.0000000825","discount":0},"name":"OpenAI: GPT-4o-mini (fast) (cheapThink)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini:fast","provider":"Azure","variant":"fast","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075","discount":0},"name":"OpenAI: GPT-4o-mini (cheapest-provider) (cheapThink)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-4o-mini-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":true,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"name":"OpenAI: GPT-4o-mini (normal) (cheapThink)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","base_model":"openai/gpt-4o-mini-normal","variant":"normal","modifier":"cheapThink","request_multiplier":2,"required_plans":null},{"id":"mistralai/mixtral-8x22b-instruct:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral: Mixtral 8x22B Instruct (fast) (cheapThink)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct:fast","provider":"Mistral","variant":"fast","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002","discount":0},"name":"Mistral: Mixtral 8x22B Instruct (cheapest-provider) (cheapThink)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct:cheapest-provider","provider":"Mistral","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"mistralai/mixtral-8x22b-instruct-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":65536,"architecture":{"input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"name":"Mistral: Mixtral 8x22B Instruct (normal) (cheapThink)","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","base_model":"mistralai/mixtral-8x22b-instruct-normal","variant":"normal","modifier":"cheapThink","request_multiplier":20,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"gryphe/mythomax-l2-13b:fast:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004","discount":0},"name":"MythoMax 13B (fast) (cheapThink)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b:fast","provider":"DeepInfra","variant":"fast","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b:cheapest-provider:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011","discount":0},"name":"MythoMax 13B (cheapest-provider) (cheapThink)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b:cheapest-provider","provider":"Parasail","variant":"cheapest-provider","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"gryphe/mythomax-l2-13b:quality:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":4096,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000004","completion":"0.0000004","discount":0},"name":"MythoMax 13B (quality) (cheapThink)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b:quality","provider":"DeepInfra","variant":"quality","modifier":"cheapThink","request_multiplier":3,"required_plans":null},{"id":"gryphe/mythomax-l2-13b-normal:cheapThink","object":"model","owned_by":"cv11","created":1791290919,"context_length":8192,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":null,"is_moderated":false,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"name":"MythoMax 13B (normal) (cheapThink)","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","base_model":"gryphe/mythomax-l2-13b-normal","variant":"normal","modifier":"cheapThink","request_multiplier":1,"required_plans":null},{"id":"perceptron/perceptron-mk1.5-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":36864,"architecture":{"input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"name":"Perceptron: Perceptron Mk1.5 (Reasoning)","description":"Perceptron Mk1.5 is Perceptron's embodied reasoning model for physical agents. It accepts text, image, video, and audio input, and answers with text plus optional structured annotations: points, boxes, polygons, tracks,...","base_model":"perceptron/perceptron-mk1.5","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":3,"required_plans":null},{"id":"fireworks/ember-1-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000003","completion":"0.000015","input_cache_read":"0.0000003"},"name":"Fireworks: Ember-1 (Reasoning)","description":"Ember-1 is a specialized reasoning model from Fireworks Research, built on [Kimi K3](https://openrouter.ai/moonshotai/kimi-k3). It is designed to make every token go further: it produces shorter reasoning traces, using roughly 40%...","base_model":"fireworks/ember-1","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"upstage/solar-mini4-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.000000005"},"name":"Upstage: Solar Mini 4 (Reasoning)","description":"Solar Mini 4 is Upstage's compact, cost-efficient language model, a 35B-parameter mixture-of-experts with 3B active parameters and a 524K context window. It is built for agentic use cases where response...","base_model":"upstage/solar-mini4","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-luna-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna Pro (Reasoning)","description":"GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-luna-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-luna-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}],"discount":0},"name":"OpenAI: GPT-6 Luna (Reasoning)","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-luna","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-6-sol-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol Pro (Reasoning)","description":"GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-sol-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-6-sol-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-6 Sol (Reasoning)","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-6-sol","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"prism-ml/ternary-bonsai-2-27b-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.0000005","input_cache_read":"0.0000000375"},"name":"PrismML: Ternary Bonsai 2 27B (Reasoning)","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","base_model":"prism-ml/ternary-bonsai-2-27b","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (Reasoning)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"inception/mercury-2.5-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":260000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000004","completion":"0.00000015","input_cache_read":"0.000000004"},"name":"Inception: Mercury 2.5 (Reasoning)","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","base_model":"inception/mercury-2.5","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-mini-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000025","completion":"0.0000001","input_cache_read":"0.0000000025"},"name":"Nex AGI: Nex-N2.5-Mini (Reasoning)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","base_model":"nex-agi/nex-n2.5-mini","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"nex-agi/nex-n2.5-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"Nex AGI: Nex-N2.5-Pro (Reasoning)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","base_model":"nex-agi/nex-n2.5-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"ibm-granite/granite-4.2-8b-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000006","completion":"0.00000025","input_cache_read":"0.000000015"},"name":"IBM: Granite 4.2 8B (Reasoning)","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","base_model":"ibm-granite/granite-4.2-8b","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"tencent/hy4-preview-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview (Reasoning)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-flash-vision-exp-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (Reasoning)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.8-27b-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.00000053125","discount":0.25},"name":"Qwen: Qwen3.8 27B (Reasoning)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"bytedance-seed/seed-2.0-code-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.000001","completion":"0.000006"}]},"name":"ByteDance Seed: Seed-2.0-Code (Reasoning)","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","base_model":"bytedance-seed/seed-2.0-code","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}],"discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813 (Reasoning)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"upstage/solar-pro4-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000036","input_cache_read":"0.000000018"},"name":"Upstage: Solar Pro 4 (Reasoning)","description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","base_model":"upstage/solar-pro4","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731 (Reasoning)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"thinkingmachines/inkling-small-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000045","completion":"0.0000012","input_cache_read":"0.0000001"},"name":"Thinking Machines: Inkling Small (Reasoning)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","base_model":"thinkingmachines/inkling-small","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"anthropic/claude-opus-5-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5 (Reasoning)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"thinkingmachines/inkling-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":524288,"architecture":{"input_modalities":["text","image","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.00000405","input_cache_read":"0.00000016"},"name":"Thinking Machines: Inkling (Reasoning)","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","base_model":"thinkingmachines/inkling","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":12,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3 (Reasoning)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-luna-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna Pro (Reasoning)","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-luna-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-luna-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Luna (Reasoning)","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-luna","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.6-terra-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra Pro (Reasoning)","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-terra-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-terra-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0},"name":"OpenAI: GPT-5.6 Terra (Reasoning)","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-terra","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":15,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol-pro-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol Pro (Reasoning)","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-sol-pro","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.6-sol-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}],"discount":0.5},"name":"OpenAI: GPT-5.6 Sol (Reasoning)","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.6-sol","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":13,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (Reasoning)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-sonnet-5-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5 (Reasoning)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra (Reasoning)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8 (Reasoning)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"google/gemini-3.1-flash-lite-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite (Reasoning)","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","provider":"Google","flex":true,"base_model":"google/gemini-3.1-flash-lite","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"x-ai/grok-4.3-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (Reasoning)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.5-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}],"discount":0},"name":"OpenAI: GPT-5.5 (Reasoning)","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.5","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":38,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3-preview-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000018","completion":"0.0000006","input_cache_read":"0.00000006"},"name":"Tencent: Hy3 preview (Reasoning)","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","base_model":"tencent/hy3-preview","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-image-2-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":272000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["image","text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","top_logprobs","verbosity"],"pricing":{"prompt":"0.000008","completion":"0.000015","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"name":"OpenAI: GPT-5.4 Image 2 (Reasoning)","description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","base_model":"openai/gpt-5.4-image-2","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":65,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7 (Reasoning)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"openai/gpt-5.4-nano-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.0000001","completion":"0.000000625","web_search":"0.01","input_cache_read":"0.00000001","discount":0},"name":"OpenAI: GPT-5.4 Nano (Reasoning)","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4-nano","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"openai/gpt-5.4-mini-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000375","completion":"0.00000225","web_search":"0.01","input_cache_read":"0.0000000375","discount":0},"name":"OpenAI: GPT-5.4 Mini (Reasoning)","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4-mini","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-super-120b-a12b-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000008","completion":"0.00000045"},"name":"NVIDIA: Nemotron 3 Super (Reasoning)","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","base_model":"nvidia/nemotron-3-super-120b-a12b","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"bytedance-seed/seed-2.0-lite-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"name":"ByteDance Seed: Seed-2.0-Lite (Reasoning)","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","base_model":"bytedance-seed/seed-2.0-lite","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.4-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1050000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000125","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000025","completion":"0.00001125","input_cache_read":"0.00000025"}],"discount":0},"name":"OpenAI: GPT-5.4 (Reasoning)","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.4","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":19,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"inception/mercury-2-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":128000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"name":"Inception: Mercury 2 (Reasoning)","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","base_model":"inception/mercury-2","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":3,"required_plans":null},{"id":"google/gemini-3.1-flash-lite-preview-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3.1 Flash Lite Preview (Reasoning)","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","provider":"Google AI Studio","flex":true,"base_model":"google/gemini-3.1-flash-lite-preview","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"bytedance-seed/seed-2.0-mini-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.0000001","completion":"0.0000004","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000002","completion":"0.0000008"}]},"name":"ByteDance Seed: Seed-2.0-Mini (Reasoning)","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","base_model":"bytedance-seed/seed-2.0-mini","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":1,"required_plans":null},{"id":"openai/gpt-5.3-codex-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex (Reasoning)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6 (Reasoning)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6 (Reasoning)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"upstage/solar-pro-3-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":131072,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"name":"Upstage: Solar Pro 3 (Reasoning)","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","base_model":"upstage/solar-pro-3","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":2,"required_plans":null},{"id":"google/gemini-3-flash-preview-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","minimal"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000416666666666667","discount":0},"name":"Google: Gemini 3 Flash Preview (Reasoning)","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","provider":"Google","flex":true,"base_model":"google/gemini-3-flash-preview","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"openai/gpt-5.2-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000875","completion":"0.000007","web_search":"0.01","input_cache_read":"0.0000000875","discount":0},"name":"OpenAI: GPT-5.2 (Reasoning)","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.2","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":16,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000000625","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000000625","discount":0},"name":"OpenAI: GPT-5.1 (Reasoning)","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","provider":"OpenAI","flex":true,"base_model":"openai/gpt-5.1","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.1-codex-mini-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low"]},"is_moderated":false,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.00000003"},"name":"OpenAI: GPT-5.1-Codex-Mini (Reasoning)","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","base_model":"openai/gpt-5.1-codex-mini","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":5,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4.1-flash:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (fast) (Reasoning)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:fast","provider":"Decart","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000018","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (cheapest-provider) (Reasoning)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:cheapest-provider","provider":"Decart","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000098","completion":"0.00000039","input_cache_read":"0.000000009","discount":0},"name":"DeepSeek: DeepSeek V4.1 Flash (quality) (Reasoning)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash:quality","provider":"Morph","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4.1-flash-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000012563","completion":"0.00000132","input_cache_read":"0.000000012563"},"name":"DeepSeek: DeepSeek V4.1 Flash (normal) (Reasoning)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","base_model":"deepseek/deepseek-v4.1-flash-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"tencent/hy4-preview:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (fast) (Reasoning)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:fast","provider":"SiliconFlow","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (cheapest-provider) (Reasoning)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","discount":0},"name":"Tencent: Hy4 preview (quality) (Reasoning)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview:quality","provider":"SiliconFlow","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"tencent/hy4-preview-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"name":"Tencent: Hy4 preview (normal) (Reasoning)","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","base_model":"tencent/hy4-preview-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-flash-vision-exp:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048575,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000014","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (fast) (Reasoning)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:fast","provider":"GMICloud","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686","discount":0.51},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (cheapest-provider) (Reasoning)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000044","completion":"0.00000132","input_cache_read":"0.000000028","discount":0},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (quality) (Reasoning)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp:quality","provider":"SiliconFlow","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":4,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-vision-exp-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"name":"DeepSeek: DeepSeek V4 Flash Vision Exp (normal) (Reasoning)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","base_model":"deepseek/deepseek-v4-flash-vision-exp-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"qwen/qwen3.8-27b:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (fast) (Reasoning)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:fast","provider":"DeepInfra","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (cheapest-provider) (Reasoning)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:cheapest-provider","provider":"Phala","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000015","completion":"0.000001875","input_cache_read":"0.0000000375","discount":0.25},"name":"Qwen: Qwen3.8 27B (quality) (Reasoning)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b:quality","provider":"DeepInfra","variant":"quality","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":4,"required_plans":null},{"id":"qwen/qwen3.8-27b-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","medium","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000425","completion":"0.00000255","input_cache_read":"0.000000085","input_cache_write":"0.00000053125"},"name":"Qwen: Qwen3.8 27B (normal) (Reasoning)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","base_model":"qwen/qwen3.8-27b-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000957","completion":"0.0000028776","input_cache_read":"0.000000099","discount":0.34},"name":"DeepSeek: DeepSeek V4 Pro 0813 (fast) (Reasoning)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:fast","provider":"Phala","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000633664","completion":"0.000001980032","input_cache_read":"0.000000075072","discount":0.36},"name":"DeepSeek: DeepSeek V4 Pro 0813 (cheapest-provider) (Reasoning)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:cheapest-provider","provider":"Ionstream","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001056","completion":"0.000003168","input_cache_read":"0.000000035","discount":0},"name":"DeepSeek: DeepSeek V4 Pro 0813 (quality) (Reasoning)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813:quality","provider":"NextBit","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":11,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-pro-0813-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}]},"name":"DeepSeek: DeepSeek V4 Pro 0813 (normal) (Reasoning)","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","base_model":"deepseek/deepseek-v4-pro-0813-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":7,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"deepseek/deepseek-v4-flash-0731:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000119","completion":"0.000000238","input_cache_read":"0.0000000238","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (fast) (Reasoning)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:fast","provider":"DigitalOcean","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1024000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000000044","completion":"0.000000132","input_cache_read":"0.0000000014","discount":0.9},"name":"DeepSeek: DeepSeek V4 Flash 0731 (cheapest-provider) (Reasoning)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:cheapest-provider","provider":"StreamLake","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.00000005","discount":0},"name":"DeepSeek: DeepSeek V4 Flash 0731 (quality) (Reasoning)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731:quality","provider":"Parasail","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":1,"required_plans":null},{"id":"deepseek/deepseek-v4-flash-0731-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.0000000118","completion":"0.00000128","input_cache_read":"0.0000000118"},"name":"DeepSeek: DeepSeek V4 Flash 0731 (normal) (Reasoning)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","base_model":"deepseek/deepseek-v4-flash-0731-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-opus-5:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002","discount":0},"name":"Anthropic: Claude Opus 5 (fast) (Reasoning)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5:fast","provider":"Anthropic","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":133,"required_plans":["ultra","enterprise"]},{"id":"anthropic/claude-opus-5:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 5 (cheapest-provider) (Reasoning)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-5-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 5 (normal) (Reasoning)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","base_model":"anthropic/claude-opus-5-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000201","completion":"0.00001005","input_cache_read":"0.000000201","discount":0.33},"name":"MoonshotAI: Kimi K3 (fast) (Reasoning)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:fast","provider":"Decart","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.000000195","discount":0.35},"name":"MoonshotAI: Kimi K3 (cheapest-provider) (Reasoning)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:cheapest-provider","provider":"Phala","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":26,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.000001274","completion":"0.000013296","input_cache_read":"0.000000278","discount":0},"name":"MoonshotAI: Kimi K3 (quality) (Reasoning)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3:quality","provider":"Morph","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":29,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"moonshotai/kimi-k3-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1048576,"architecture":{"input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"prompt":"0.00000095","completion":"0.000014","input_cache_read":"0.00000031"},"name":"MoonshotAI: Kimi K3 (normal) (Reasoning)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","base_model":"moonshotai/kimi-k3-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":28,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"tencent/hy3:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000013","completion":"0.00000053","input_cache_read":"0.000000033","discount":0},"name":"Tencent: Hy3 (fast) (Reasoning)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:fast","provider":"DeepInfra","variant":"fast","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","discount":0,"overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (cheapest-provider) (Reasoning)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:cheapest-provider","provider":"Tencent","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.00000014","completion":"0.00000058","input_cache_read":"0.000000035","discount":0},"name":"Tencent: Hy3 (quality) (Reasoning)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3:quality","provider":"GMICloud","variant":"quality","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"tencent/hy3-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","low","none"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"name":"Tencent: Hy3 (normal) (Reasoning)","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","base_model":"tencent/hy3-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"low","request_multiplier":2,"required_plans":null},{"id":"anthropic/claude-sonnet-5:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (fast) (Reasoning)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5:fast","provider":"Amazon Bedrock","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004","discount":0},"name":"Anthropic: Claude Sonnet 5 (cheapest-provider) (Reasoning)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-5-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"name":"Anthropic: Claude Sonnet 5 (normal) (Reasoning)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","base_model":"anthropic/claude-sonnet-5-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":27,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (fast) (Reasoning)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:fast","provider":"DeepInfra","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (cheapest-provider) (Reasoning)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:cheapest-provider","provider":"DeepInfra","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b:quality-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":256000,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.000000625","completion":"0.000003125","input_cache_read":"0.0000001875","discount":0},"name":"NVIDIA: Nemotron 3 Ultra (quality) (Reasoning)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b:quality","provider":"Venice","variant":"quality","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":8,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"nvidia/nemotron-3-ultra-550b-a55b-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":262144,"architecture":{"input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium"]},"is_moderated":false,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"name":"NVIDIA: Nemotron 3 Ultra (normal) (Reasoning)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","base_model":"nvidia/nemotron-3-ultra-550b-a55b-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":6,"required_plans":["basic","plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (fast) (Reasoning)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.8 (cheapest-provider) (Reasoning)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.8-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.8 (normal) (Reasoning)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","base_model":"anthropic/claude-opus-4.8-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (fast) (Reasoning)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3:fast","provider":"xAI","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","discount":0,"overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (cheapest-provider) (Reasoning)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3:cheapest-provider","provider":"xAI","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"x-ai/grok-4.3-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["high","medium","low","none"]},"is_moderated":false,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"name":"SpaceXAI: Grok 4.3 (normal) (Reasoning)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","base_model":"x-ai/grok-4.3-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":10,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (fast) (Reasoning)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.7 (cheapest-provider) (Reasoning)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.7-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.7 (normal) (Reasoning)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","base_model":"anthropic/claude-opus-4.7-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (fast) (Reasoning)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex:fast","provider":"Azure","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175","discount":0},"name":"OpenAI: GPT-5.3-Codex (cheapest-provider) (Reasoning)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex:cheapest-provider","provider":"Azure","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"openai/gpt-5.3-codex-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":400000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high","medium","low","none"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"name":"OpenAI: GPT-5.3-Codex (normal) (Reasoning)","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","base_model":"openai/gpt-5.3-codex-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":32,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (fast) (Reasoning)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6:fast","provider":"Claude Platform on AWS","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","discount":0},"name":"Anthropic: Claude Sonnet 4.6 (cheapest-provider) (Reasoning)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-sonnet-4.6-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"name":"Anthropic: Claude Sonnet 4.6 (normal) (Reasoning)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","base_model":"anthropic/claude-sonnet-4.6-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":40,"required_plans":["plus","pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:fast-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (fast) (Reasoning)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6:fast","provider":"Google","variant":"fast","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6:cheapest-provider-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001","discount":0},"name":"Anthropic: Claude Opus 4.6 (cheapest-provider) (Reasoning)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6:cheapest-provider","provider":"Claude Platform on AWS","variant":"cheapest-provider","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]},{"id":"anthropic/claude-opus-4.6-normal-reasoning","object":"model","owned_by":"cv11","created":1791290919,"context_length":1000000,"architecture":{"input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"]},"is_moderated":true,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"name":"Anthropic: Claude Opus 4.6 (normal) (Reasoning)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","base_model":"anthropic/claude-opus-4.6-normal","variant":"normal","modifier":"reasoning","default_reasoning_effort":"medium","request_multiplier":67,"required_plans":["pro","ultra","enterprise"]}]}