{"data":[{"id":"gpt-5.6-sol","name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.704912500000001,"max_completion_cost":4.172736,"max_cost":9.1821925},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8776.3428933608,"max_completion_cost":6419.267944858183,"max_cost":14125.732847409283},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.5","name":"xAI: Grok 4.5","created":1783523154,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":1.08665,"max_completion_cost":3.2599500000000003,"max_cost":3.2599500000000003},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":1671.684360640152,"max_completion_cost":5015.053081920457,"max_cost":5015.053081920457},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.5-20260708","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":2.5750000000000003e-06,"max_prompt_cost":2.1733,"max_completion_cost":1.3909120000000001,"max_cost":3.2860296},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0039613373474885125,"max_prompt_cost":3343.368721280304,"max_completion_cost":2139.755981619395,"max_cost":5055.173506575819},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0105845e-06,"completion":3.2599500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7755140000000002e-07,"input_cache_write":0.0,"max_prompt_cost":1.034838528,"max_completion_cost":0.4172736000000001,"max_cost":1.322757312},"sats_pricing":{"prompt":0.0015546664553953413,"completion":0.005015053081920457,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002731421327840279,"input_cache_write":0.0,"max_prompt_cost":1591.9784503248297,"max_completion_cost":641.9267944858185,"max_cost":2034.9079385200441},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5","name":"Anthropic: Claude Fable 5","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":5.43325e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.0299999999999999e-06,"input_cache_write":1.2875000000000001e-05,"max_prompt_cost":10.866500000000002,"max_completion_cost":6.95456,"max_cost":16.430148000000003},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.0835842180320076,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0015845349389954045,"input_cache_write":0.019806686737442562,"max_prompt_cost":16716.843606401522,"max_completion_cost":10698.779908096973,"max_cost":25275.867532879103},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8","name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":2.716625e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.433250000000001,"max_completion_cost":3.47728,"max_cost":8.215074000000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.0417921090160038,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8358.421803200761,"max_completion_cost":5349.389954048486,"max_cost":12637.933766439552},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.25995e-05,"completion":0.00019559700000000002,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-06,"input_cache_write":0.0,"max_prompt_cost":34.229475,"max_completion_cost":25.036416000000003,"max_cost":55.093155},"sats_pricing":{"prompt":0.05015053081920456,"completion":0.3009031849152274,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.004753604816986215,"input_cache_write":0.0,"max_prompt_cost":52658.057360164785,"max_completion_cost":38515.6076691491,"max_cost":84754.3970844557},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":1.3583125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.0600000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.086932,"max_completion_cost":0.17386400000000002,"max_cost":0.23297776000000003},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.00208960545080019,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":3.16906987799081e-05,"input_cache_write":0.0,"max_prompt_cost":133.73474885121215,"max_completion_cost":267.46949770242435,"max_cost":358.4091269212486},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.149875000000001e-07,"completion":4.889925e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":7.725e-08,"input_cache_write":0.0,"max_prompt_cost":0.32599500000000003,"max_completion_cost":0.6259104,"max_cost":0.847587},"sats_pricing":{"prompt":0.0012537632704801142,"completion":0.007522579622880683,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00011884012042465535,"input_cache_write":0.0,"max_prompt_cost":501.5053081920456,"max_completion_cost":962.8901917287275,"max_cost":1303.9138012993185},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-chat","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":1.52131e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.8025e-07,"input_cache_write":0.0,"max_prompt_cost":0.2434096,"max_completion_cost":0.2492514304,"max_cost":0.4615046016},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.02340358104896213,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002772936143241958,"input_cache_write":0.0,"max_prompt_cost":374.45729678339404,"max_completion_cost":383.4442719061955,"max_cost":709.9710347013151},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.3-chat-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-codex","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":1.52131e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.8025e-07,"input_cache_write":0.0,"max_prompt_cost":0.760655,"max_completion_cost":1.9472768,"max_cost":2.4645222},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.02340358104896213,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002772936143241958,"input_cache_write":0.0,"max_prompt_cost":1170.1790524481064,"max_completion_cost":2995.6583742671523,"max_cost":3791.380129931865},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.3-codex-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.605e-07,"completion":2.1630000000000003e-06,"request":0.0,"image":5.149999999999999e-07,"web_search":0.01442,"internal_reasoning":3.09e-06,"input_cache_read":5.15e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.378011648,"max_completion_cost":0.14175220500000002,"max_cost":0.4961384855},"sats_pricing":{"prompt":0.0005545872286483916,"completion":0.0033275233718903503,"request":0.001,"image":0.0007922674694977023,"web_search":22.183489145935667,"internal_reasoning":0.004753604816986215,"input_cache_read":7.922674694977023e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":581.5268578672159,"max_completion_cost":218.0692441768341,"max_cost":763.2512280145777},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":5.433250000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":1.2875000000000002e-06,"max_prompt_cost":0.21732999999999997,"max_completion_cost":0.34772800000000004,"max_cost":0.4955124},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.00835842180320076,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0019806686737442562,"max_prompt_cost":334.33687212803034,"max_completion_cost":534.9389954048487,"max_cost":762.2880684519093},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":1.2875000000000002e-06,"max_prompt_cost":1.1409824999999998,"max_completion_cost":0.8345472000000002,"max_cost":1.8364384999999999},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0019806686737442562,"max_prompt_cost":1755.2685786721593,"max_completion_cost":1283.853588971637,"max_cost":2825.1465694818567},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna","name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":1.2875000000000002e-06,"max_prompt_cost":1.1409824999999998,"max_completion_cost":0.8345472000000002,"max_cost":1.8364384999999999},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0019806686737442562,"max_prompt_cost":1755.2685786721593,"max_completion_cost":1283.853588971637,"max_cost":2825.1465694818567},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":3.2187500000000003e-06,"max_prompt_cost":2.8524562500000004,"max_completion_cost":2.086368,"max_cost":4.59109625},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.004951671684360641,"max_prompt_cost":4388.1714466804,"max_completion_cost":3209.6339724290915,"max_cost":7062.866423704641},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra","name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":3.2187500000000003e-06,"max_prompt_cost":2.8524562500000004,"max_completion_cost":2.086368,"max_cost":4.59109625},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.004951671684360641,"max_prompt_cost":4388.1714466804,"max_completion_cost":3209.6339724290915,"max_cost":7062.866423704641},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.704912500000001,"max_completion_cost":4.172736,"max_cost":9.1821925},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8776.3428933608,"max_completion_cost":6419.267944858183,"max_cost":14125.732847409283},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-latest","name":"xAI: Grok Latest","created":1783519360,"description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":1.08665,"max_completion_cost":3.2599500000000003,"max_cost":3.2599500000000003},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":1671.684360640152,"max_completion_cost":5015.053081920457,"max_cost":5015.053081920457},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~x-ai/grok-latest","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","created":1783443096,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.60655e-07,"completion":1.52131e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.854e-07,"input_cache_write":0.0,"max_prompt_cost":0.09970057216,"max_completion_cost":0.04985028608,"max_cost":0.1246257152},"sats_pricing":{"prompt":0.0011701790524481063,"completion":0.0023403581048962127,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002852162890191729,"input_cache_write":0.0,"max_prompt_cost":153.3777087624782,"max_completion_cost":76.6888543812391,"max_cost":191.72213595309776},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-3.0-mini-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0","name":"AionLabs: Aion-3.0","created":1783443095,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.725e-07,"input_cache_write":0.0,"max_prompt_cost":0.42728816640000006,"max_completion_cost":0.21364408320000003,"max_cost":0.5341102080000001},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0011884012042465537,"input_cache_write":0.0,"max_prompt_cost":657.3330375534781,"max_completion_cost":328.66651877673905,"max_cost":821.6662969418476},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-3.0-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"hy3","name":"Tencent: Hy3","created":1783344048,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.52131e-07,"completion":6.30257e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.605e-08,"input_cache_write":0.0,"max_prompt_cost":0.039880228864,"max_completion_cost":0.165218091008,"max_cost":0.165218091008},"sats_pricing":{"prompt":0.00023403581048962128,"completion":0.0009695769291712882,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.545872286483917e-05,"input_cache_write":0.0,"max_prompt_cost":61.35108350499128,"max_completion_cost":254.16877452067817,"max_cost":254.16877452067817},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hy3-20260706","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","created":1783002429,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.519899999999999e-08,"completion":1.3039799999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":0.0,"max_prompt_cost":0.017091526655999997,"max_completion_cost":0.004272881663999999,"max_cost":0.019227967487999997},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0002006021232768182,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0,"max_prompt_cost":26.293321502139115,"max_completion_cost":6.573330375534779,"max_cost":29.579986689906505},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"poolside/laguna-xs-2.1-20260625","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","created":1782837225,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.8025e-07,"completion":1.0815000000000001e-06,"request":0.0,"image":0.0,"web_search":0.01442,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.011812864,"max_completion_cost":0.07087718400000001,"max_cost":0.07087718400000001},"sats_pricing":{"prompt":0.0002772936143241958,"completion":0.0016637616859451752,"request":0.001,"image":0.0,"web_search":22.183489145935667,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.172714308350496,"max_completion_cost":109.036285850103,"max_cost":109.036285850103},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":66000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","created":1782312964,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.716625e-08,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5750000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.02848587776,"max_cost":0.02848587776},"sats_pricing":{"prompt":4.17921090160038e-05,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.961337347488512e-06,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":43.8222025035652,"max_cost":43.8222025035652},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nex-agi/nex-n2-mini","alias_ids":null,"forwarded_model_id":null},{"id":"fugu-ultra","name":"Sakana: Fugu Ultra","created":1782276303,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":5.433250000000001,"max_completion_cost":4.172736,"max_cost":8.910530000000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":8358.421803200761,"max_completion_cost":6419.267944858183,"max_cost":13707.811757249248},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sakana/fugu-ultra-20260615","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","created":1781754065,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.605e-07,"completion":2.1630000000000003e-06,"request":0.0,"image":0.0,"web_search":0.01442,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.047251456,"max_completion_cost":0.07087718400000001,"max_cost":0.106315776},"sats_pricing":{"prompt":0.0005545872286483916,"completion":0.0033275233718903503,"request":0.001,"image":0.0,"web_search":22.183489145935667,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.69085723340199,"max_completion_cost":109.036285850103,"max_cost":163.5544287751545},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","created":1781754054,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.442e-06,"completion":8.652000000000001e-06,"request":0.0,"image":2.0599999999999998e-06,"web_search":0.01442,"internal_reasoning":1.236e-05,"input_cache_read":2.06e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":0.094502912,"max_completion_cost":0.28350873600000004,"max_cost":0.33076019200000006},"sats_pricing":{"prompt":0.0022183489145935664,"completion":0.013310093487561401,"request":0.001,"image":0.003169069877990809,"web_search":22.183489145935667,"internal_reasoning":0.01901441926794486,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0005942006021232768,"max_prompt_cost":145.38171446680397,"max_completion_cost":436.145143400412,"max_cost":508.836000633814},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-pro-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"north-mini-code","name":"North Mini Code","created":1781723748,"description":"PPQ.AI model","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08345472,"max_completion_cost":0.27818239999999994,"max_cost":0.27818239999999994},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":128.38535889716366,"max_completion_cost":427.95119632387883,"max_cost":427.95119632387883},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code","name":"MoonshotAI: Kimi K2.7 Code","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.8130135e-07,"completion":3.7924085000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.5347e-07,"input_cache_write":0.0,"max_prompt_cost":0.2048134610944,"max_completion_cost":0.9941571338240001,"max_cost":0.9941571338240001},"sats_pricing":{"prompt":0.0012019410553002691,"completion":0.005834178418634131,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00023609570591031533,"input_cache_write":0.0,"max_prompt_cost":315.08163600063375,"max_completion_cost":1529.3948673744255,"max_cost":1529.3948673744255},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-latest","name":"Anthropic: Claude Fable Latest","created":1781029944,"description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":5.43325e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.0299999999999999e-06,"input_cache_write":1.2875000000000001e-05,"max_prompt_cost":10.866500000000002,"max_completion_cost":6.95456,"max_cost":16.430148000000003},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.0835842180320076,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0015845349389954045,"input_cache_write":0.019806686737442562,"max_prompt_cost":16716.843606401522,"max_completion_cost":10698.779908096973,"max_cost":25275.867532879103},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-fable-latest","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","created":1780937140,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.07121469439999999,"max_completion_cost":0.28485877759999995,"max_cost":0.28485877759999995},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":109.55550625891298,"max_completion_cost":438.22202503565194,"max_cost":438.22202503565194},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nex-agi/nex-n2-pro","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","created":1780581864,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04172736,"max_completion_cost":0.13909119999999997,"max_cost":0.13909119999999997},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":64.19267944858183,"max_completion_cost":213.97559816193942,"max_cost":213.97559816193942},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.433249999999999e-07,"completion":2.3906300000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":0.0,"max_prompt_cost":0.14242938879999997,"max_completion_cost":0.03916808192000001,"max_cost":0.17269563391999998},"sats_pricing":{"prompt":0.0008358421803200759,"completion":0.003677705593408335,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0,"max_prompt_cost":219.11101251782597,"max_completion_cost":60.25552844240216,"max_cost":265.672102677864},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.47728e-07,"completion":1.390912e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.592000000000001e-08,"input_cache_write":4.12e-07,"max_prompt_cost":0.347728,"max_completion_cost":0.091154808832,"max_cost":0.416094106624},"sats_pricing":{"prompt":0.0005349389954048486,"completion":0.0021397559816193944,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010141023609570591,"input_cache_write":0.0006338139755981618,"max_prompt_cost":534.9389954048486,"max_completion_cost":140.23104801140863,"max_cost":640.1122814134051},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.7-plus-20260602","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.18e-08,"input_cache_write":0.0,"max_prompt_cost":0.325995,"max_completion_cost":0.17091526656,"max_cost":0.45418144992},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.507209633972428e-05,"input_cache_write":0.0,"max_prompt_cost":501.50530819204556,"max_completion_cost":262.9332150213912,"max_cost":698.705219458089},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.7-flash","name":"StepFun: Step 3.7 Flash","created":1779985069,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":256000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":1.2496475e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.1200000000000005e-08,"input_cache_write":0.0,"max_prompt_cost":0.05563648,"max_completion_cost":0.31990976,"max_cost":0.31990976},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0019224370147361749,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.33813975598162e-05,"input_cache_write":0.0,"max_prompt_cost":85.59023926477579,"max_completion_cost":492.1438757724607,"max_cost":492.1438757724607},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":256000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"stepfun/step-3.7-flash-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","created":1779913703,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":5.43325e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.0299999999999999e-06,"input_cache_write":1.2875000000000001e-05,"max_prompt_cost":10.866500000000002,"max_completion_cost":6.95456,"max_cost":16.430148000000003},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.0835842180320076,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0015845349389954045,"input_cache_write":0.019806686737442562,"max_prompt_cost":16716.843606401522,"max_completion_cost":10698.779908096973,"max_cost":25275.867532879103},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.8-opus-fast-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-max","name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":4.0749375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":1.6093750000000002e-06,"max_prompt_cost":1.3583125000000003,"max_completion_cost":0.267055104,"max_cost":1.5363492360000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.00626881635240057,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.0024758358421803203,"max_prompt_cost":2089.6054508001903,"max_completion_cost":410.83314847092373,"max_cost":2363.494216447473},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.7-max-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"grok-build-0.1","name":"xAI: Grok Build 0.1","created":1779298123,"description":"Grok Build 0.1 is xAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":0.27818239999999994,"max_completion_cost":0.5563647999999999,"max_cost":0.5563647999999999},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":427.95119632387883,"max_completion_cost":855.9023926477577,"max_cost":855.9023926477577},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-build-0.1-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash","name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.0815000000000001e-06,"completion":6.489e-06,"request":0.0,"image":1.545e-06,"web_search":0.01442,"internal_reasoning":9.270000000000001e-06,"input_cache_read":1.545e-07,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":1.1340349440000002,"max_completion_cost":0.425263104,"max_cost":1.488420864},"sats_pricing":{"prompt":0.0016637616859451752,"completion":0.00998257011567105,"request":0.001,"image":0.0023768024084931073,"web_search":22.183489145935667,"internal_reasoning":0.014260814450958644,"input_cache_read":0.0002376802408493107,"input_cache_write":0.00013204457824961708,"max_prompt_cost":1744.580573601648,"max_completion_cost":654.217715100618,"max_cost":2289.762002852163},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.25995e-05,"completion":0.0001629975,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-06,"input_cache_write":3.8625e-05,"max_prompt_cost":32.5995,"max_completion_cost":20.86368,"max_cost":49.290443999999994},"sats_pricing":{"prompt":0.05015053081920456,"completion":0.25075265409602276,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.004753604816986215,"input_cache_write":0.05942006021232767,"max_prompt_cost":50150.530819204556,"max_completion_cost":32096.339724290916,"max_cost":75827.60259863728},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"perceptron-mk1","name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":1.6299750000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00534110208,"max_completion_cost":0.013352755200000002,"max_cost":0.017358581760000002},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0025075265409602284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.216662969418476,"max_completion_cost":20.54165742354619,"max_cost":26.704154650610047},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perceptron/perceptron-mk1-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.149875e-08,"completion":6.791562500000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-08,"input_cache_write":0.0,"max_prompt_cost":0.02136440832,"max_completion_cost":0.04450918400000001,"max_cost":0.06053249024000001},"sats_pricing":{"prompt":0.0001253763270480114,"completion":0.001044802725400095,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.376802408493107e-05,"input_cache_write":0.0,"max_prompt_cost":32.8666518776739,"max_completion_cost":68.47219141182063,"max_cost":93.12218032007605},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ring-2.6-1t-20260508","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.8025e-07,"completion":1.0815000000000001e-06,"request":0.0,"image":2.5749999999999997e-07,"web_search":0.01442,"internal_reasoning":1.545e-06,"input_cache_read":2.575e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.189005824,"max_completion_cost":0.07087718400000001,"max_cost":0.248070144},"sats_pricing":{"prompt":0.0002772936143241958,"completion":0.0016637616859451752,"request":0.001,"image":0.0003961337347488511,"web_search":22.183489145935667,"internal_reasoning":0.0023768024084931073,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":290.76342893360794,"max_completion_cost":109.036285850103,"max_cost":381.62700047536043},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-chat-latest","name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":2.1733000000000002,"max_completion_cost":4.172736,"max_cost":5.65058},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":3343.3687212803043,"max_completion_cost":6419.267944858183,"max_cost":8692.75867532879},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-chat-latest-20260505","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.3","name":"xAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":1.3583125000000003,"max_completion_cost":2.7166250000000005,"max_cost":2.7166250000000005},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":2089.6054508001903,"max_completion_cost":4179.210901600381,"max_cost":4179.210901600381},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.3-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.43325e-08,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.01424293888,"max_cost":0.01424293888},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.922674694977023e-05,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":21.9111012517826,"max_cost":21.9111012517826},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.6299750000000002e-06,"completion":8.149875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.42728816640000006,"max_completion_cost":2.136440832,"max_cost":2.136440832},"sats_pricing":{"prompt":0.0025075265409602284,"completion":0.01253763270480114,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":657.3330375534781,"max_completion_cost":3286.66518776739,"max_cost":3286.66518776739},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3.5-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni","created":1777393095,"description":"PPQ.AI model","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08345472,"max_completion_cost":0.27818239999999994,"max_cost":0.27818239999999994},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":128.38535889716366,"max_completion_cost":427.95119632387883,"max_cost":427.95119632387883},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"laguna-m.1","name":"Poolside: Laguna M.1","created":1777388504,"description":"Laguna M.1 is the flagship coding agent model from [Poolside](https://poolside.ai/), optimized for complex software engineering tasks. Designed for agentic coding workflows, it supports tool calling and reasoning, with a 256K...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.01424293888,"max_cost":0.06409322496},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":21.9111012517826,"max_cost":98.5999556330217},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"poolside/laguna-m.1-20260312","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-latest","name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":5.433250000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":1.2875000000000002e-06,"max_prompt_cost":0.21732999999999997,"max_completion_cost":0.34772800000000004,"max_cost":0.4955124},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.00835842180320076,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0019806686737442562,"max_prompt_cost":334.33687212803034,"max_completion_cost":534.9389954048487,"max_cost":762.2880684519093},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-haiku-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-mini-latest","name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":8.149875000000001e-07,"completion":4.889925e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":7.725e-08,"input_cache_write":0.0,"max_prompt_cost":0.32599500000000003,"max_completion_cost":0.6259104,"max_cost":0.847587},"sats_pricing":{"prompt":0.0012537632704801142,"completion":0.007522579622880683,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00011884012042465535,"input_cache_write":0.0,"max_prompt_cost":501.5053081920456,"max_completion_cost":962.8901917287275,"max_cost":1303.9138012993185},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~openai/gpt-mini-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-pro-latest","name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.442e-06,"completion":8.652000000000001e-06,"request":0.0,"image":2.0599999999999998e-06,"web_search":0.01442,"internal_reasoning":1.236e-05,"input_cache_read":2.06e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":1.512046592,"max_completion_cost":0.5670174720000001,"max_cost":1.984561152},"sats_pricing":{"prompt":0.0022183489145935664,"completion":0.013310093487561401,"request":0.001,"image":0.003169069877990809,"web_search":22.183489145935667,"internal_reasoning":0.01901441926794486,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0005942006021232768,"max_prompt_cost":2326.1074314688635,"max_completion_cost":872.290286800824,"max_cost":3053.0160038028835},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~google/gemini-pro-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-latest","name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":7.17189e-07,"completion":3.7054765000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-07,"input_cache_write":0.0,"max_prompt_cost":0.188006793216,"max_completion_cost":0.9713684316160001,"max_cost":0.9713684316160001},"sats_pricing":{"prompt":0.0011033116780225004,"completion":0.005700443669782919,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002376802408493107,"input_cache_write":0.0,"max_prompt_cost":289.22653652353034,"max_completion_cost":1494.3371053715734,"max_cost":1494.3371053715734},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~moonshotai/kimi-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-flash-latest","name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0815000000000001e-06,"completion":6.489e-06,"request":0.0,"image":1.545e-06,"web_search":0.01442,"internal_reasoning":9.270000000000001e-06,"input_cache_read":1.545e-07,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":1.1340349440000002,"max_completion_cost":0.425263104,"max_cost":1.488420864},"sats_pricing":{"prompt":0.0016637616859451752,"completion":0.00998257011567105,"request":0.001,"image":0.0023768024084931073,"web_search":22.183489145935667,"internal_reasoning":0.014260814450958644,"input_cache_read":0.0002376802408493107,"input_cache_write":0.00013204457824961708,"max_prompt_cost":1744.580573601648,"max_completion_cost":654.217715100618,"max_cost":2289.762002852163},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~google/gemini-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":2.5750000000000003e-06,"max_prompt_cost":2.1733,"max_completion_cost":1.3909120000000001,"max_cost":3.2860296},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0039613373474885125,"max_prompt_cost":3343.368721280304,"max_completion_cost":2139.755981619395,"max_cost":5055.173506575819},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-sonnet-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-latest","name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.704912500000001,"max_completion_cost":4.172736,"max_cost":9.1821925},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8776.3428933608,"max_completion_cost":6419.267944858183,"max_cost":14125.732847409283},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~openai/gpt-latest","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.95597e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":3.8625e-07,"max_prompt_cost":0.325995,"max_completion_cost":0.12818644992,"max_cost":0.4328170416},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0030090318491522734,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0005942006021232768,"max_prompt_cost":501.50530819204556,"max_completion_cost":197.1999112660434,"max_cost":665.8385675804151},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-plus-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-flash","name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.0374687500000003e-07,"completion":1.22248125e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.4140625e-07,"max_prompt_cost":0.20374687500000002,"max_completion_cost":0.0801165312,"max_cost":0.27051065100000005},"sats_pricing":{"prompt":0.00031344081762002855,"completion":0.0018806449057201708,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.000371375376327048,"max_prompt_cost":313.44081762002855,"max_completion_cost":123.24994454127712,"max_cost":416.1491047377595},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.52131e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.039880228864,"max_completion_cost":0.28485877759999995,"max_cost":0.28485877759999995},"sats_pricing":{"prompt":0.00023403581048962128,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":61.35108350499128,"max_completion_cost":438.22202503565194,"max_cost":438.22202503565194},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-max-preview","name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.130116e-06,"completion":6.780696e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":1.339e-06,"max_prompt_cost":0.296253128704,"max_completion_cost":0.444379693056,"max_cost":0.666569539584},"sats_pricing":{"prompt":0.0017385517350657581,"completion":0.01043131041039455,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.002059895420694026,"max_prompt_cost":455.7509060370781,"max_completion_cost":683.6263590556172,"max_cost":1025.4395385834257},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-max-preview-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-27b","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.0969525000000005e-07,"completion":2.60796e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04059237580800001,"max_completion_cost":0.34183053312,"max_cost":0.34183053312},"sats_pricing":{"prompt":0.0004764300427824434,"completion":0.004012042465536365,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":62.44663856758042,"max_completion_cost":525.8664300427824,"max_cost":525.8664300427824},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-27b-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":5.704912500000001,"max_completion_cost":4.172736,"max_cost":9.1821925},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":8776.3428933608,"max_completion_cost":6419.267944858183,"max_cost":14125.732847409283},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":4.7269275e-07,"completion":9.453855e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.73375e-09,"input_cache_write":0.0,"max_prompt_cost":0.495654273024,"max_completion_cost":0.363028032,"max_cost":0.6771682890240001},"sats_pricing":{"prompt":0.0007271826968784661,"completion":0.0014543653937569322,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.743939153858343e-06,"input_cache_write":0.0,"max_prompt_cost":762.5063235620345,"max_completion_cost":558.4763112026619,"max_cost":1041.7444791633654},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v4-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":8.367205e-08,"completion":1.673441e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.854e-08,"input_cache_write":0.0,"max_prompt_cost":0.0877365035008,"max_completion_cost":0.0109670629376,"max_cost":0.0932200349696},"sats_pricing":{"prompt":0.00012871969576929172,"completion":0.00025743939153858344,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.8521628901917285e-05,"input_cache_write":0.0,"max_prompt_cost":134.97238371098084,"max_completion_cost":16.871547963872604,"max_cost":143.40815769291711},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v4-flash-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.149875e-08,"completion":6.791562500000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-08,"input_cache_write":0.0,"max_prompt_cost":0.02136440832,"max_completion_cost":0.022254592000000004,"max_cost":0.04094844928000001},"sats_pricing":{"prompt":0.0001253763270480114,"completion":0.001044802725400095,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.376802408493107e-05,"input_cache_write":0.0,"max_prompt_cost":32.8666518776739,"max_completion_cost":34.236095705910316,"max_cost":62.994416098874986},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ling-2.6-1t-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.845895e-08,"completion":2.281965e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.163e-08,"input_cache_write":0.0,"max_prompt_cost":0.0179461029888,"max_completion_cost":0.059820343296,"max_cost":0.059820343296},"sats_pricing":{"prompt":0.00010531611472032957,"completion":0.00035105371573443194,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.32752337189035e-05,"input_cache_write":0.0,"max_prompt_cost":27.607987577246075,"max_completion_cost":92.02662525748693,"max_cost":92.02662525748693},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hy3-preview-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.7269275e-07,"completion":9.453855e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.708e-09,"input_cache_write":0.0,"max_prompt_cost":0.495654273024,"max_completion_cost":0.123913568256,"max_cost":0.557611057152},"sats_pricing":{"prompt":0.0007271826968784661,"completion":0.0014543653937569322,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.704325780383457e-06,"input_cache_write":0.0,"max_prompt_cost":762.5063235620345,"max_completion_cost":190.62658089050862,"max_cost":857.8196140072888},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1048576,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1409825e-07,"completion":3.04262e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.884e-08,"input_cache_write":0.0,"max_prompt_cost":0.029910171648,"max_completion_cost":0.079760457728,"max_cost":0.079760457728},"sats_pricing":{"prompt":0.00017552685786721597,"completion":0.00046807162097924255,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.436697829187133e-05,"input_cache_write":0.0,"max_prompt_cost":46.01331262874346,"max_completion_cost":122.70216700998256,"max_cost":122.70216700998256},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"xiaomi/mimo-v2.5-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.693199999999998e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.0599999999999998e-06,"input_cache_write":0.0,"max_prompt_cost":2.3645503999999997,"max_completion_cost":2.086368,"max_cost":3.3381887999999993},"sats_pricing":{"prompt":0.013373474885121214,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.003169069877990809,"input_cache_write":0.0,"max_prompt_cost":3637.5851687529703,"max_completion_cost":3209.6339724290915,"max_cost":5135.414355886545},"per_request_limits":null,"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-image-2-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-08,"completion":3.2599499999999994e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0600000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.002848587776,"max_completion_cost":0.0010682204159999998,"max_cost":0.00356073472},"sats_pricing":{"prompt":1.671684360640152e-05,"completion":5.015053081920455e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.1690698779908098e-06,"input_cache_write":0.0,"max_prompt_cost":4.38222025035652,"max_completion_cost":1.6433325938836947,"max_cost":5.47777531294565},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ling-2.6-flash-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-latest","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":2.716625e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.433250000000001,"max_completion_cost":3.47728,"max_cost":8.215074000000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.0417921090160038,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8358.421803200761,"max_completion_cost":5349.389954048486,"max_cost":12637.933766439552},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-opus-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.17189e-07,"completion":3.7054765000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-07,"input_cache_write":0.0,"max_prompt_cost":0.188006793216,"max_completion_cost":0.9713684316160001,"max_cost":0.9713684316160001},"sats_pricing":{"prompt":0.0011033116780225004,"completion":0.005700443669782919,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002376802408493107,"input_cache_write":0.0,"max_prompt_cost":289.22653652353034,"max_completion_cost":1494.3371053715734,"max_cost":1494.3371053715734},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.6-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7","name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":2.716625e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.433250000000001,"max_completion_cost":3.47728,"max_cost":8.215074000000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.0417921090160038,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8358.421803200761,"max_completion_cost":5349.389954048486,"max_cost":12637.933766439552},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0497039e-06,"completion":3.2990694e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.8478200000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.20994078000000002,"max_completion_cost":0.42228088319999996,"max_cost":0.497859564},"sats_pricing":{"prompt":0.0016148470923783868,"completion":0.005075233718903501,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002842655680557756,"input_cache_write":0.0,"max_prompt_cost":322.9694184756774,"max_completion_cost":649.6299160196481,"max_cost":765.898906670892},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5.1-20260406","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":6.519899999999999e-08,"completion":3.585945e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017091526655999997,"max_completion_cost":0.094003396608,"max_cost":0.094003396608},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0005516558390112502,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":26.293321502139115,"max_completion_cost":144.61326826176517,"max_cost":144.61326826176517},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-31b-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":1.3039799999999997e-07,"completion":3.803275e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.27e-08,"input_cache_write":0.0,"max_prompt_cost":0.03418305331199999,"max_completion_cost":0.09970057216,"max_cost":0.09970057216},"sats_pricing":{"prompt":0.0002006021232768182,"completion":0.0005850895262240532,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00014260814450958644,"input_cache_write":0.0,"max_prompt_cost":52.58664300427823,"max_completion_cost":153.3777087624782,"max_cost":153.3777087624782},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-4-31b-it-20260402","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-plus","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.5316125e-07,"completion":2.1189675e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.1843749999999996e-07,"max_prompt_cost":0.35316125,"max_completion_cost":0.13886865408,"max_cost":0.46888512839999996},"sats_pricing":{"prompt":0.0005432974172080493,"completion":0.003259784503248296,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0006437173189668831,"max_prompt_cost":543.2974172080494,"max_completion_cost":213.63323720488032,"max_cost":721.325114878783},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-plus-04-02","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.30398e-06,"completion":4.346599999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.472e-07,"input_cache_write":0.0,"max_prompt_cost":0.26438455296,"max_completion_cost":0.5697175551999999,"max_cost":0.6631868415999999},"sats_pricing":{"prompt":0.0020060212327681825,"completion":0.006686737442560607,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00038028838535889714,"input_cache_write":0.0,"max_prompt_cost":406.72481698621453,"max_completion_cost":876.4440500713039,"max_cost":1020.2356520361271},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5v-turbo-20260401","alias_ids":null,"forwarded_model_id":null},{"id":"trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.18e-08,"input_cache_write":0.0,"max_prompt_cost":0.07121469439999999,"max_completion_cost":0.0695456,"max_cost":0.11902729439999998},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.507209633972428e-05,"input_cache_write":0.0,"max_prompt_cost":109.55550625891298,"max_completion_cost":106.98779908096972,"max_cost":183.10961812707967},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"arcee-ai/trinity-large-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20-multi-agent","name":"xAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of xAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":2.7166250000000005,"max_completion_cost":5.433250000000001,"max_cost":5.433250000000001},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":4179.210901600381,"max_completion_cost":8358.421803200761,"max_cost":8358.421803200761},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20","name":"xAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from xAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":2.7166250000000005,"max_completion_cost":5.433250000000001,"max_cost":5.433250000000001},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":4179.210901600381,"max_completion_cost":8358.421803200761,"max_cost":8358.421803200761},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.20-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","created":1774907286,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","created":1774907255,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.18e-08,"input_cache_write":0.0,"max_prompt_cost":0.08345472,"max_completion_cost":0.1043184,"max_cost":0.16169352},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.507209633972428e-05,"input_cache_write":0.0,"max_prompt_cost":128.38535889716366,"max_completion_cost":160.4816986214546,"max_cost":248.74663286325463},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","alias_ids":null,"forwarded_model_id":null},{"id":"reka-edge","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00178036736,"max_completion_cost":0.00178036736,"max_cost":0.00178036736},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.738887656472825,"max_completion_cost":2.738887656472825,"max_cost":2.738887656472825},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"rekaai/reka-edge-2603","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.6079599999999995e-07,"completion":1.0431839999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051274579967999986,"max_completion_cost":0.20509831987199995,"max_cost":0.20509831987199995},"sats_pricing":{"prompt":0.0004012042465536364,"completion":0.0016048169862145456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":78.87996450641734,"max_completion_cost":315.51985802566935,"max_cost":315.51985802566935},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.7-20260318","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-08,"input_cache_write":0.0,"max_prompt_cost":0.04272881664,"max_completion_cost":0.17091526656,"max_cost":0.17091526656},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.376802408493107e-05,"input_cache_write":0.0,"max_prompt_cost":65.7333037553478,"max_completion_cost":262.9332150213912,"max_cost":262.9332150213912},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-2603","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-turbo","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.30398e-06,"completion":4.346599999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.472e-07,"input_cache_write":0.0,"max_prompt_cost":0.34183053312,"max_completion_cost":0.5697175551999999,"max_cost":0.7406328217599999},"sats_pricing":{"prompt":0.0020060212327681825,"completion":0.006686737442560607,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00038028838535889714,"input_cache_write":0.0,"max_prompt_cost":525.8664300427824,"max_completion_cost":876.4440500713039,"max_cost":1139.377265092695},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5-turbo-20260315","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.6932e-08,"completion":4.889925e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.022788702208,"max_completion_cost":0.12818644992,"max_cost":0.12818644992},"sats_pricing":{"prompt":0.00013373474885121215,"completion":0.0007522579622880683,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":35.05776200285216,"max_completion_cost":197.1999112660434,"max_cost":197.1999112660434},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07121469439999999,"max_completion_cost":0.28485877759999995,"max_cost":0.3204661247999999},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":109.55550625891298,"max_completion_cost":438.22202503565194,"max_cost":492.99977816510835},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-9b","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":1.629975e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02848587776,"max_completion_cost":0.04272881664,"max_cost":0.04272881664},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0002507526540960228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.8222025035652,"max_completion_cost":65.7333037553478,"max_cost":65.7333037553478},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-9b-20260310","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.25995e-05,"completion":0.00019559700000000002,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-06,"input_cache_write":0.0,"max_prompt_cost":34.229475,"max_completion_cost":25.036416000000003,"max_cost":55.093155},"sats_pricing":{"prompt":0.05015053081920456,"completion":0.3009031849152274,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.004753604816986215,"input_cache_write":0.0,"max_prompt_cost":52658.057360164785,"max_completion_cost":38515.6076691491,"max_cost":84754.3970844557},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":2.8524562500000004,"max_completion_cost":2.086368,"max_cost":4.59109625},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.0,"max_prompt_cost":4388.1714466804,"max_completion_cost":3209.6339724290915,"max_cost":7062.866423704641},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"mercury-2","name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":8.149875000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.03477279999999999,"max_completion_cost":0.040749375000000004,"max_cost":0.061939049999999995},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0012537632704801142,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":53.493899540484854,"max_completion_cost":62.6881635240057,"max_cost":95.28600855648865},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inception/mercury-2-20260304","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.8025e-07,"completion":1.0815000000000001e-06,"request":0.0,"image":2.5749999999999997e-07,"web_search":0.01442,"internal_reasoning":1.545e-06,"input_cache_read":2.575e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.189005824,"max_completion_cost":0.07087718400000001,"max_cost":0.248070144},"sats_pricing":{"prompt":0.0002772936143241958,"completion":0.0016637616859451752,"request":0.001,"image":0.0003961337347488511,"web_search":22.183489145935667,"internal_reasoning":0.0023768024084931073,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":290.76342893360794,"max_completion_cost":109.036285850103,"max_cost":381.62700047536043},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02848587776,"max_completion_cost":0.05697175552,"max_cost":0.0712146944},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.8222025035652,"max_completion_cost":87.6444050071304,"max_cost":109.555506258913},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.605e-07,"completion":2.1630000000000003e-06,"request":0.0,"image":0.0,"web_search":0.01442,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.047251456,"max_completion_cost":0.07087718400000001,"max_cost":0.106315776},"sats_pricing":{"prompt":0.0005545872286483916,"completion":0.0033275233718903503,"request":0.001,"image":0.0,"web_search":22.183489145935667,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.69085723340199,"max_completion_cost":109.036285850103,"max_cost":163.5544287751545},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.52131e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.039880228864,"max_completion_cost":0.08901836799999999,"max_cost":0.116436025344},"sats_pricing":{"prompt":0.00023403581048962128,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.922674694977023e-05,"input_cache_write":0.0,"max_prompt_cost":61.35108350499128,"max_completion_cost":136.94438282364123,"max_cost":179.12325273332274},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-27b","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1189675e-07,"completion":1.695174e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.055547461632,"max_completion_cost":0.111094923264,"max_cost":0.152755519488},"sats_pricing":{"prompt":0.00032597845032482966,"completion":0.0026078276025986373,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":85.45329488195215,"max_completion_cost":170.9065897639043,"max_cost":234.99656092536839},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-27b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":2.260232e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.074063282176,"max_completion_cost":0.592506257408,"max_cost":0.592506257408},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.0034771034701315162,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":113.93772650926952,"max_completion_cost":911.5018120741562,"max_cost":911.5018120741562},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.063225e-08,"completion":2.82529e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07063225,"max_completion_cost":0.018515820544,"max_cost":0.084519115408},"sats_pricing":{"prompt":0.00010865948344160988,"completion":0.00043463793376643953,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":108.65948344160986,"max_completion_cost":28.48443162731738,"max_cost":130.0228071620979},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-flash-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048756,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.442e-06,"completion":8.652000000000001e-06,"request":0.0,"image":2.0599999999999998e-06,"web_search":0.01442,"internal_reasoning":1.236e-05,"input_cache_read":2.06e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":1.512046592,"max_completion_cost":0.5670174720000001,"max_cost":1.984561152},"sats_pricing":{"prompt":0.0022183489145935664,"completion":0.013310093487561401,"request":0.001,"image":0.003169069877990809,"web_search":22.183489145935667,"internal_reasoning":0.01901441926794486,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0005942006021232768,"max_prompt_cost":2326.1074314688635,"max_completion_cost":872.290286800824,"max_cost":3053.0160038028835},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"aion-2.0","name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.6932e-07,"completion":1.73864e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":0.11394351104,"max_completion_cost":0.05697175552,"max_cost":0.1424293888},"sats_pricing":{"prompt":0.0013373474885121216,"completion":0.0026746949770242432,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":175.2888100142608,"max_completion_cost":87.6444050071304,"max_cost":219.111012517826},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-2.0-20260223","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.442e-06,"completion":8.652000000000001e-06,"request":0.0,"image":2.0599999999999998e-06,"web_search":0.01442,"internal_reasoning":1.236e-05,"input_cache_read":2.06e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":1.512046592,"max_completion_cost":0.5670174720000001,"max_cost":1.984561152},"sats_pricing":{"prompt":0.0022183489145935664,"completion":0.013310093487561401,"request":0.001,"image":0.003169069877990809,"web_search":22.183489145935667,"internal_reasoning":0.01901441926794486,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0005942006021232768,"max_prompt_cost":2326.1074314688635,"max_completion_cost":872.290286800824,"max_cost":3053.0160038028835},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-07,"input_cache_write":3.8625e-06,"max_prompt_cost":3.2599500000000003,"max_completion_cost":2.086368,"max_cost":4.9290444},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0004753604816986214,"input_cache_write":0.005942006021232768,"max_prompt_cost":5015.053081920457,"max_completion_cost":3209.6339724290915,"max_cost":7582.76025986373},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":1.695174e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.282529,"max_completion_cost":0.111094923264,"max_cost":0.37510810272},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.0026078276025986373,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":434.63793376643946,"max_completion_cost":170.9065897639043,"max_cost":577.0600919030264},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-plus-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":256000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.1836025000000006e-07,"completion":2.6622925000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1433e-07,"input_cache_write":0.0,"max_prompt_cost":0.05483531468800001,"max_completion_cost":0.34895200256000003,"max_cost":0.34895200256000003},"sats_pricing":{"prompt":0.0006435984788464586,"completion":0.004095626683568373,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017588337822848992,"input_cache_write":0.0,"max_prompt_cost":84.35773981936302,"max_completion_cost":536.8219806686737,"max_cost":536.8219806686737},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":9.77985e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.15e-08,"input_cache_write":3.8625e-07,"max_prompt_cost":0.03204661248,"max_completion_cost":0.19227967488,"max_cost":0.19227967488},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0015045159245761367,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.922674694977023e-05,"input_cache_write":0.0005942006021232768,"max_prompt_cost":49.29997781651085,"max_completion_cost":295.7998668990651,"max_cost":295.7998668990651},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":2.0863679999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.236e-07,"input_cache_write":0.0,"max_prompt_cost":0.13219227648,"max_completion_cost":0.4230152847359999,"max_cost":0.4230152847359999},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.003209633972429091,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00019014419267944857,"input_cache_write":0.0,"max_prompt_cost":203.36240849310727,"max_completion_cost":650.7597071779431,"max_cost":650.7597071779431},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":8.47587e-07,"completion":4.237935e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.222189846528,"max_completion_cost":0.13886865408,"max_cost":0.333284769792},"sats_pricing":{"prompt":0.0013039138012993186,"completion":0.006519569006496592,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":341.8131795278086,"max_completion_cost":213.63323720488032,"max_cost":512.7197692917129},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-max-thinking-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":2.716625e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":5.433250000000001,"max_completion_cost":3.47728,"max_cost":8.215074000000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.0417921090160038,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":8358.421803200761,"max_completion_cost":5349.389954048486,"max_cost":12637.933766439552},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.195315e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.21e-08,"input_cache_write":0.0,"max_prompt_cost":0.031334465536,"max_completion_cost":0.22788702208,"max_cost":0.22788702208},"sats_pricing":{"prompt":0.0001838852796704167,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011091744572967834,"input_cache_write":0.0,"max_prompt_cost":48.20442275392172,"max_completion_cost":350.5776200285216,"max_cost":350.5776200285216},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.5-flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":3.25995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02848587776,"max_completion_cost":0.02136440832,"max_cost":0.04272881664},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0005015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.8222025035652,"max_completion_cost":32.8666518776739,"max_cost":65.7333037553478},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"stepfun/step-3.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.0749375000000006e-07,"completion":2.20046625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0909e-07,"input_cache_write":0.0,"max_prompt_cost":0.10682204160000001,"max_completion_cost":0.57683902464,"max_cost":0.57683902464},"sats_pricing":{"prompt":0.0006268816352400571,"completion":0.0033851608302963077,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00032166059261606715,"input_cache_write":0.0,"max_prompt_cost":164.33325938836953,"max_completion_cost":887.3996006971953,"max_cost":887.3996006971953},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.5-0127","alias_ids":null,"forwarded_model_id":null},{"id":"solar-pro-3","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-08,"input_cache_write":0.0,"max_prompt_cost":0.02086368,"max_completion_cost":0.08345472,"max_cost":0.08345472},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.376802408493107e-05,"input_cache_write":0.0,"max_prompt_cost":32.096339724290914,"max_completion_cost":128.38535889716366,"max_cost":128.38535889716366},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"upstage/solar-pro-3","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2-her","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":0.0,"max_prompt_cost":0.02136440832,"max_completion_cost":0.00267055104,"max_cost":0.0233673216},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0,"max_prompt_cost":32.8666518776739,"max_completion_cost":4.108331484709238,"max_cost":35.94790049120583},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2-her-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"palmyra-x5","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6780696,"max_completion_cost":0.05341102080000001,"max_cost":0.72613951872},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1043.1310410394549,"max_completion_cost":82.16662969418476,"max_cost":1117.081007764221},"per_request_limits":null,"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"writer/palmyra-x5-20250428","alias_ids":null,"forwarded_model_id":null},{"id":"lfm-2.5-1.2b-thinking","name":"LFM2.5-1.2B-Thinking","created":1768927527,"description":"PPQ.AI model","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01068220416,"max_completion_cost":0.03560734719999999,"max_cost":0.03560734719999999},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.43332593883695,"max_completion_cost":54.77775312945649,"max_cost":54.77775312945649},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"lfm-2.5-1.2b-instruct","name":"LFM2.5-1.2B-Instruct","created":1768927521,"description":"PPQ.AI model","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01068220416,"max_completion_cost":0.03560734719999999,"max_cost":0.03560734719999999},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.43332593883695,"max_completion_cost":54.77775312945649,"max_cost":54.77775312945649},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.17803673600000003,"max_cost":0.4812555520000001},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":273.8887656472825,"max_cost":740.3555696403106},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-audio","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio-mini","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":2.60796e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08345472,"max_completion_cost":0.04272881664,"max_cost":0.11550133248},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.004012042465536365,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":128.38535889716366,"max_completion_cost":65.7333037553478,"max_cost":177.6853367136745},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-audio-mini","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7-flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.519899999999999e-08,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0300000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.013219227647999997,"max_completion_cost":0.00712146944,"max_cost":0.019272476671999998},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.584534938995405e-05,"input_cache_write":0.0,"max_prompt_cost":20.336240849310723,"max_completion_cost":10.9555506258913,"max_cost":29.648458881318327},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.7-flash-20260119","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-codex","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":1.52131e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.8025e-07,"input_cache_write":0.0,"max_prompt_cost":0.760655,"max_completion_cost":1.9472768,"max_cost":2.4645222},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.02340358104896213,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002772936143241958,"input_cache_write":0.0,"max_prompt_cost":1170.1790524481064,"max_completion_cost":2995.6583742671523,"max_cost":3791.380129931865},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-codex-20260114","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.149875e-08,"completion":3.25995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02136440832,"max_completion_cost":0.01068220416,"max_cost":0.02937606144},"sats_pricing":{"prompt":0.0001253763270480114,"completion":0.0005015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.8666518776739,"max_completion_cost":16.43332593883695,"max_cost":45.191646331801614},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07121469439999999,"max_completion_cost":0.07121469439999999,"max_cost":0.13352755199999997},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":109.55550625891298,"max_completion_cost":109.55550625891298,"max_cost":205.4165742354618},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-1.6-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":3.8625e-07,"max_prompt_cost":0.066763776,"max_completion_cost":0.17091526656,"max_cost":0.19495022592},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0005942006021232768,"max_prompt_cost":102.70828711773093,"max_completion_cost":262.9332150213912,"max_cost":299.90819838377433},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":1.9016375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.240000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.08812818432,"max_completion_cost":0.2492514304,"max_cost":0.2804078592},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.002925447631120266,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001267627951196324,"input_cache_write":0.0,"max_prompt_cost":135.57493899540484,"max_completion_cost":383.4442719061955,"max_cost":431.37480589446994},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.7-20251222","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.43325e-08,"completion":2.1733e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01424293888,"max_completion_cost":0.04955124,"max_cost":0.051406368880000004},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0003343368721280304,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.9111012517826,"max_completion_cost":76.22880684519093,"max_cost":79.0827063856758},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":228000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":1.52131e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.8025e-07,"input_cache_write":0.0,"max_prompt_cost":0.2434096,"max_completion_cost":0.2492514304,"max_cost":0.4615046016},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.02340358104896213,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002772936143241958,"input_cache_write":0.0,"max_prompt_cost":374.45729678339404,"max_completion_cost":383.4442719061955,"max_cost":709.9710347013151},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-chat-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.281965e-05,"completion":0.0001825572,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.12786,"max_completion_cost":23.3673216,"max_cost":29.5742664},"sats_pricing":{"prompt":0.03510537157344319,"completion":0.28084297258754554,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14042.148629377276,"max_completion_cost":35947.90049120583,"max_cost":45496.56155918237},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":1.52131e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.8025e-07,"input_cache_write":0.0,"max_prompt_cost":0.760655,"max_completion_cost":1.9472768,"max_cost":2.4645222},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.02340358104896213,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002772936143241958,"input_cache_write":0.0,"max_prompt_cost":1170.1790524481064,"max_completion_cost":2995.6583742671523,"max_cost":3791.380129931865},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"devstral-2512","name":"Mistral: Devstral 2 2512","created":1765285419,"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","context_length":262144,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.1200000000000005e-08,"input_cache_write":0.0,"max_prompt_cost":0.11394351104,"max_completion_cost":0.5697175551999999,"max_cost":0.5697175551999999},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.33813975598162e-05,"input_cache_write":0.0,"max_prompt_cost":175.2888100142608,"max_completion_cost":876.4440500713039,"max_cost":876.4440500713039},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/devstral-2512","alias_ids":null,"forwarded_model_id":null},{"id":"relace-search","name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":3.2599500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27818239999999994,"max_completion_cost":0.4172736000000001,"max_cost":0.5563648000000001},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.005015053081920457,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":427.95119632387883,"max_completion_cost":641.9267944858185,"max_cost":855.902392647758},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"relace/relace-search-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6v","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":9.77985e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.665000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.04272881664,"max_completion_cost":0.03204661248,"max_cost":0.06409322496},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0015045159245761367,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.714942164474728e-05,"input_cache_write":0.0,"max_prompt_cost":65.7333037553478,"max_completion_cost":49.29997781651085,"max_cost":98.5999556330217},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.6-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2874999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.5433250000000001,"max_completion_cost":1.3909120000000001,"max_cost":1.7603730000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0,"max_prompt_cost":835.8421803200761,"max_completion_cost":2139.755981619395,"max_cost":2708.1286642370465},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-max-20251204","alias_ids":null,"forwarded_model_id":null},{"id":"nova-2-lite-v1","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.325995,"max_completion_cost":0.17803401937500002,"max_cost":0.48266493705},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":501.50530819204556,"max_completion_cost":273.8845864363809,"max_cost":742.5237442560608},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-2-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":2.1733e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0600000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.05697175552,"max_cost":0.05697175552},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0003343368721280304,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.16906987799081e-05,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":87.6444050071304,"max_cost":87.6444050071304},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-14b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":1.629975e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-08,"input_cache_write":0.0,"max_prompt_cost":0.04272881664,"max_completion_cost":0.04272881664,"max_cost":0.04272881664},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0002507526540960228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.376802408493107e-05,"input_cache_write":0.0,"max_prompt_cost":65.7333037553478,"max_completion_cost":65.7333037553478,"max_cost":65.7333037553478},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-8b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0300000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.01424293888,"max_completion_cost":0.01424293888,"max_cost":0.01424293888},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.584534938995405e-05,"input_cache_write":0.0,"max_prompt_cost":21.9111012517826,"max_completion_cost":21.9111012517826,"max_cost":21.9111012517826},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-3b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2512","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.433249999999999e-07,"completion":1.6299750000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.15e-08,"input_cache_write":0.0,"max_prompt_cost":0.14242938879999997,"max_completion_cost":0.42728816640000006,"max_cost":0.42728816640000006},"sats_pricing":{"prompt":0.0008358421803200759,"completion":0.0025075265409602284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.922674694977023e-05,"input_cache_write":0.0,"max_prompt_cost":219.11101251782597,"max_completion_cost":657.3330375534781,"max_cost":657.3330375534781},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large-2512","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.3308642500000003e-07,"completion":3.496296375e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2093500000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.029835062400000004,"max_completion_cost":0.0223762968,"max_cost":0.037293828},"sats_pricing":{"prompt":0.00035857629535731265,"completion":0.0005378644430359689,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.398827444145143e-05,"input_cache_write":0.0,"max_prompt_cost":45.89776580573602,"max_completion_cost":34.42332435430201,"max_cost":57.37220725717002},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.2-20251201","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":2.716625e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":6.437500000000001e-06,"max_prompt_cost":1.0866500000000001,"max_completion_cost":1.73864,"max_cost":2.477562},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.0417921090160038,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.009903343368721281,"max_prompt_cost":1671.6843606401521,"max_completion_cost":2674.694977024243,"max_cost":3811.440342259546},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":5.433249999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01068220416,"max_completion_cost":0.03560734719999999,"max_cost":0.03560734719999999},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0008358421803200759,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.43332593883695,"max_completion_cost":54.77775312945649,"max_cost":54.77775312945649},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"allenai/olmo-3-32b-think-20251121","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.442e-06,"completion":8.652000000000001e-06,"request":0.0,"image":2.0599999999999998e-06,"web_search":0.01442,"internal_reasoning":1.236e-05,"input_cache_read":2.06e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":0.094502912,"max_completion_cost":0.28350873600000004,"max_cost":0.33076019200000006},"sats_pricing":{"prompt":0.0022183489145935664,"completion":0.013310093487561401,"request":0.001,"image":0.003169069877990809,"web_search":22.183489145935667,"internal_reasoning":0.01901441926794486,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0005942006021232768,"max_prompt_cost":145.38171446680397,"max_completion_cost":436.145143400412,"max_cost":508.836000633814},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-pro-image-preview-20251120","alias_ids":null,"forwarded_model_id":null},{"id":"cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.3583125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.17386400000000002,"max_completion_cost":0.17386400000000002,"max_cost":0.17386400000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.00208960545080019,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":267.46949770242435,"max_completion_cost":267.46949770242435,"max_cost":267.46949770242435},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2874999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.5433250000000001,"max_completion_cost":1.3909120000000001,"max_cost":1.7603730000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0,"max_prompt_cost":835.8421803200761,"max_completion_cost":2139.755981619395,"max_cost":2708.1286642370465},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-chat","name":"OpenAI: GPT-5.1 Chat","created":1763060302,"description":"GPT-5.1 Chat (AKA Instant is the fast, lightweight member of the 5.1 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.339e-07,"input_cache_write":0.0,"max_prompt_cost":0.17386400000000002,"max_completion_cost":0.34772800000000004,"max_cost":0.47812600000000005},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002059895420694026,"input_cache_write":0.0,"max_prompt_cost":267.46949770242435,"max_completion_cost":534.9389954048487,"max_cost":735.541118681667},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-chat-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.339e-07,"input_cache_write":0.0,"max_prompt_cost":0.5433250000000001,"max_completion_cost":1.3909120000000001,"max_cost":1.7603730000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0002059895420694026,"input_cache_write":0.0,"max_prompt_cost":835.8421803200761,"max_completion_cost":2139.755981619395,"max_cost":2708.1286642370465},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.10866499999999998,"max_completion_cost":0.21732999999999997,"max_cost":0.29882875},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":167.16843606401517,"max_completion_cost":334.33687212803034,"max_cost":459.71319917604177},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-07,"input_cache_write":0.0,"max_prompt_cost":0.17091526656,"max_completion_cost":0.272618752,"max_cost":0.37810551808000004},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002376802408493107,"input_cache_write":0.0,"max_prompt_cost":262.9332150213912,"max_completion_cost":419.39217239740134,"max_cost":581.6712660434163},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","alias_ids":null,"forwarded_model_id":null},{"id":"nova-premier-v1","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.3583125e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.437500000000001e-07,"input_cache_write":0.0,"max_prompt_cost":2.7166250000000005,"max_completion_cost":0.43466,"max_cost":3.0643530000000005},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.0208960545080019,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0009903343368721281,"input_cache_write":0.0,"max_prompt_cost":4179.210901600381,"max_completion_cost":668.6737442560608,"max_cost":4714.14989700523},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-premier-v1","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro-search","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.018539999999999997,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6519900000000001,"max_completion_cost":0.130398,"max_cost":0.7563084000000001},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":28.521628901917282,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1003.0106163840912,"max_completion_cost":200.60212327681822,"max_cost":1163.4923150055458},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-pro-search","alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":3.25995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0300000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.00347728,"max_completion_cost":0.01043184,"max_cost":0.01043184},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0005015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.584534938995405e-05,"input_cache_write":0.0,"max_prompt_cost":5.349389954048487,"max_completion_cost":16.048169862145457,"max_cost":16.048169862145457},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/voxtral-small-24b-2507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.149875e-08,"completion":3.25995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.8625e-08,"input_cache_write":0.0,"max_prompt_cost":0.01068220416,"max_completion_cost":0.02136440832,"max_cost":0.0267055104},"sats_pricing":{"prompt":0.0001253763270480114,"completion":0.0005015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.9420060212327674e-05,"input_cache_write":0.0,"max_prompt_cost":16.43332593883695,"max_completion_cost":32.8666518776739,"max_cost":41.083314847092375},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-oss-safeguard-20b","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-nano-12b-v2-vl","name":"Nemotron Nano 12B 2 VL","created":1761675565,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04172736,"max_completion_cost":0.13909119999999997,"max_cost":0.13909119999999997},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":64.19267944858183,"max_completion_cost":213.97559816193942,"max_cost":213.97559816193942},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7709575e-07,"completion":1.108383e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":3.8625e-07,"max_prompt_cost":0.0567492096,"max_completion_cost":0.145277976576,"max_cost":0.165707692032},"sats_pricing":{"prompt":0.00042627951196323873,"completion":0.001705118047852955,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0005942006021232768,"max_prompt_cost":87.3020440500713,"max_completion_cost":223.4932327681825,"max_cost":254.92196862620818},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.130116e-07,"completion":4.520464e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0148126564352,"max_completion_cost":0.0148126564352,"max_cost":0.025922148761599997},"sats_pricing":{"prompt":0.0001738551735065758,"completion":0.0006954206940263032,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":22.787545301853903,"max_completion_cost":22.787545301853903,"max_cost":39.878204278244326},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.847305e-08,"completion":1.217048e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0024199695499999997,"max_completion_cost":0.0159433288,"max_cost":0.0159433288},"sats_pricing":{"prompt":2.841863413088258e-05,"completion":0.00018722864839169703,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.7228410711456177,"max_completion_cost":24.52695293931231,"max_cost":24.52695293931231},"per_request_limits":null,"top_provider":{"context_length":131000,"max_completion_tokens":131000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ibm-granite/granite-4.0-h-micro","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":1.0866500000000001,"max_completion_cost":0.27818239999999994,"max_cost":1.0171044},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.0,"max_prompt_cost":1671.6843606401521,"max_completion_cost":427.95119632387883,"max_cost":1564.6965615591823},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-image-mini","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.2713805e-07,"completion":1.48327725e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0166642384896,"max_completion_cost":0.048604028928,"max_cost":0.061102207795199995},"sats_pricing":{"prompt":0.0001955870701948978,"completion":0.002281849152273807,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25.635988464585644,"max_completion_cost":74.77163302170811,"max_cost":93.99862437014734},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-8b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.2713805e-07,"completion":4.9442575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0166642384896,"max_completion_cost":0.016201342976,"max_cost":0.0286995218432},"sats_pricing":{"prompt":0.0001955870701948978,"completion":0.0007606163840912692,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25.635988464585644,"max_completion_cost":24.92387767390271,"max_cost":44.15086902234194},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2875000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":4.3466000000000005,"max_completion_cost":1.3909120000000001,"max_cost":4.3466000000000005},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0019806686737442562,"input_cache_write":0.0,"max_prompt_cost":6686.7374425606085,"max_completion_cost":2139.755981619395,"max_cost":6686.7374425606085},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-image","alias_ids":null,"forwarded_model_id":null},{"id":"o3-deep-research","name":"OpenAI: o3 Deep Research","created":1760129661,"description":"o3-deep-research is OpenAI's advanced model for deep research, designed to tackle complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":4.346600000000001e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.5750000000000003e-06,"input_cache_write":0.0,"max_prompt_cost":2.1733000000000002,"max_completion_cost":4.3466000000000005,"max_cost":5.433250000000001},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.06686737442560609,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0039613373474885125,"input_cache_write":0.0,"max_prompt_cost":3343.3687212803043,"max_completion_cost":6686.7374425606085,"max_cost":8358.421803200761},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-deep-research-2025-06-26","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-deep-research","name":"OpenAI: o4 Mini Deep Research","created":1760129642,"description":"o4-mini-deep-research is OpenAI's faster, more affordable deep research model—ideal for tackling complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":0.43465999999999994,"max_completion_cost":0.8693199999999999,"max_cost":1.08665},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":668.6737442560607,"max_completion_cost":1337.3474885121213,"max_cost":1671.684360640152},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o4-mini-deep-research-2025-06-26","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-nemotron-super-49b-v1.5","name":"NVIDIA: Llama 3.3 Nemotron Super 49B V1.5","created":1760101395,"description":"Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.00712146944,"max_cost":0.05697175552},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":10.9555506258913,"max_cost":87.6444050071304},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/llama-3.3-nemotron-super-49b-v1.5","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.163e-07,"completion":1.8025000000000001e-06,"request":0.0,"image":3.09e-07,"web_search":0.01442,"internal_reasoning":2.5750000000000003e-06,"input_cache_read":3.09e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.0070877184,"max_completion_cost":0.059064320000000003,"max_cost":0.059064320000000003},"sats_pricing":{"prompt":0.000332752337189035,"completion":0.0027729361432419584,"request":0.001,"image":0.0004753604816986214,"web_search":22.183489145935667,"internal_reasoning":0.0039613373474885125,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":10.9036285850103,"max_completion_cost":90.86357154175249,"max_cost":90.86357154175249},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash-image","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.412645e-07,"completion":1.695174e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018515820544,"max_completion_cost":0.055547461632,"max_cost":0.06943432704},"sats_pricing":{"prompt":0.00021731896688321976,"completion":0.0026078276025986373,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.48443162731738,"max_completion_cost":85.45329488195215,"max_cost":106.81661860244016},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.412645e-07,"completion":5.65058e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018515820544,"max_completion_cost":0.018515820544,"max_cost":0.032402685952},"sats_pricing":{"prompt":0.00021731896688321976,"completion":0.0008692758675328791,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.48443162731738,"max_completion_cost":28.48443162731738,"max_cost":49.84775534780542},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.629975e-05,"completion":0.000130398,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.5199,"max_completion_cost":16.690944,"max_cost":21.124475999999998},"sats_pricing":{"prompt":0.02507526540960228,"completion":0.20060212327681823,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10030.106163840912,"max_completion_cost":25677.071779432732,"max_cost":32497.543970844552},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.672595e-07,"completion":1.9016375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.240000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.092517381,"max_completion_cost":0.0311564288,"max_cost":0.116018230152},"sats_pricing":{"prompt":0.0007188242750752654,"completion":0.002925447631120266,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001267627951196324,"input_cache_write":0.0,"max_prompt_cost":142.32720646490253,"max_completion_cost":47.93053398827444,"max_cost":178.48052353034382},"per_request_limits":null,"top_provider":{"context_length":198000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.6","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-07,"input_cache_write":3.8625e-06,"max_prompt_cost":3.2599500000000003,"max_completion_cost":1.043184,"max_cost":4.0944972},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0004753604816986214,"input_cache_write":0.005942006021232768,"max_prompt_cost":5015.053081920457,"max_completion_cost":1604.8169862145458,"max_cost":6298.906670892093},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.933955e-07,"completion":4.455265e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04806991872,"max_completion_cost":0.029198024704,"max_cost":0.058039975936},"sats_pricing":{"prompt":0.000451354777372841,"completion":0.0006853905878624623,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":73.94996672476627,"max_completion_cost":44.91775756615433,"max_cost":89.2877376010141},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.2-exp","alias_ids":null,"forwarded_model_id":null},{"id":"cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":5.433249999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.545e-07,"input_cache_write":0.0,"max_prompt_cost":0.04272881664,"max_completion_cost":0.07121469439999999,"max_cost":0.07121469439999999},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0008358421803200759,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002376802408493107,"input_cache_write":0.0,"max_prompt_cost":65.7333037553478,"max_completion_cost":109.55550625891298,"max_cost":109.55550625891298},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/cydonia-24b-v4.1","alias_ids":null,"forwarded_model_id":null},{"id":"relace-apply-3","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.236525e-07,"completion":1.3583125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.23645504,"max_completion_cost":0.17386400000000002,"max_cost":0.29209152000000005},"sats_pricing":{"prompt":0.0014209317065441293,"completion":0.00208960545080019,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":363.75851687529706,"max_completion_cost":267.46949770242435,"max_cost":449.3487561400729},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"relace/relace-apply-3","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":2.82529e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.037031641088,"max_completion_cost":0.09257910272,"max_cost":0.120352833536},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.004346379337664395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":56.96886325463476,"max_completion_cost":142.4221581365869,"max_cost":185.14880557756297},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":9.56252e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1330000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.015667232768,"max_cost":0.069078253568},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0014710822373633337,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017429884328949455,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":24.10221137696086,"max_cost":106.26884107114562},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.47587e-07,"completion":4.237935e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6068e-07,"input_cache_write":1.00425e-06,"max_prompt_cost":0.222189846528,"max_completion_cost":0.13886865408,"max_cost":0.333284769792},"sats_pricing":{"prompt":0.0013039138012993186,"completion":0.006519569006496592,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00024718745048328317,"input_cache_write":0.0015449215655205196,"max_prompt_cost":341.8131795278086,"max_completion_cost":213.63323720488032,"max_cost":512.7197692917129},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-max","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.063225e-07,"completion":3.5316125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.339e-07,"input_cache_write":8.368749999999999e-07,"max_prompt_cost":0.7063225,"max_completion_cost":0.23144775680000002,"max_cost":0.89148070544},"sats_pricing":{"prompt":0.0010865948344160987,"completion":0.005432974172080494,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002059895420694026,"input_cache_write":0.0012874346379337662,"max_prompt_cost":1086.5948344160988,"max_completion_cost":356.05539534146726,"max_cost":1371.4391506892725},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-plus","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-codex","name":"OpenAI: GPT-5 Codex","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2874999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.5433250000000001,"max_completion_cost":1.3909120000000001,"max_cost":1.7603730000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0,"max_prompt_cost":835.8421803200761,"max_completion_cost":2139.755981619395,"max_cost":2708.1286642370465},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-codex","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.933955e-07,"completion":1.0323175e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.339e-07,"input_cache_write":0.0,"max_prompt_cost":0.04806991872,"max_completion_cost":0.03382697984,"max_cost":0.072282914816},"sats_pricing":{"prompt":0.000451354777372841,"completion":0.0015881001426081443,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002059895420694026,"input_cache_write":0.0,"max_prompt_cost":73.94996672476627,"max_completion_cost":52.03886547298367,"max_cost":111.19883885279668},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.1-terminus","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1189675e-07,"completion":1.05948375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.017e-08,"input_cache_write":2.510625e-07,"max_prompt_cost":0.21189675,"max_completion_cost":0.06943432704,"max_cost":0.267444211632},"sats_pricing":{"prompt":0.00032597845032482966,"completion":0.001629892251624148,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.179686262082079e-05,"input_cache_write":0.0003862303913801299,"max_prompt_cost":325.9784503248296,"max_completion_cost":106.81661860244016,"max_cost":411.4317452067818},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.05948375e-07,"completion":8.47587e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.013886865408,"max_completion_cost":0.027773730816,"max_cost":0.038188879872},"sats_pricing":{"prompt":0.00016298922516241483,"completion":0.0013039138012993186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.363323720488037,"max_completion_cost":42.72664744097607,"max_cost":58.749140231342096},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.779850000000001e-08,"completion":1.1953150000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.025637289984000004,"max_completion_cost":0.019584040960000004,"max_cost":0.043619000320000004},"sats_pricing":{"prompt":0.0001504515924576137,"completion":0.0018388527967041675,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":39.43998225320868,"max_completion_cost":30.12776422120108,"max_cost":67.10274758358422},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":8.47587e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":3.3475e-07,"max_prompt_cost":0.282529,"max_completion_cost":0.027773730816,"max_cost":0.301044820544},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.0013039138012993186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0005149738551735065,"max_prompt_cost":434.63793376643946,"max_completion_cost":42.72664744097607,"max_cost":463.1223653937569},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":8.47587e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.282529,"max_completion_cost":0.027773730816,"max_cost":0.301044820544},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.0013039138012993186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":434.63793376643946,"max_completion_cost":42.72664744097607,"max_cost":463.1223653937569},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-nano-9b-v2","name":"Nemotron Nano 9B V2","created":1757106807,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04172736,"max_completion_cost":0.13909119999999997,"max_cost":0.13909119999999997},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":64.19267944858183,"max_completion_cost":213.97559816193942,"max_cost":213.97559816193942},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.17091526656,"max_completion_cost":0.272618752,"max_cost":0.37810551808000004},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":262.9332150213912,"max_completion_cost":419.39217239740134,"max_cost":581.6712660434163},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2-0905","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.412645e-07,"completion":1.695174e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01157238784,"max_completion_cost":0.055547461632,"max_cost":0.062490894336},"sats_pricing":{"prompt":0.00021731896688321976,"completion":0.0026078276025986373,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17.80276976707336,"max_completion_cost":85.45329488195215,"max_cost":96.13495674219615},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-70b","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":1.412645e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.018515820544,"max_completion_cost":0.05697175552,"max_cost":0.05697175552},"sats_pricing":{"prompt":0.00021731896688321976,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":28.48443162731738,"max_completion_cost":87.6444050071304,"max_cost":87.6444050071304},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-4-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-405b","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":3.2599500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14242938879999997,"max_completion_cost":0.42728816640000006,"max_cost":0.42728816640000006},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.005015053081920457,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":219.11101251782597,"max_completion_cost":657.3330375534781,"max_cost":657.3330375534781},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-4-405b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.281965e-07,"completion":8.584535000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.339e-07,"input_cache_write":0.0,"max_prompt_cost":0.03738771456,"max_completion_cost":0.028129804288000004,"max_cost":0.05803997593600001},"sats_pricing":{"prompt":0.00035105371573443194,"completion":0.0013206306449057201,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002059895420694026,"input_cache_write":0.0,"max_prompt_cost":57.516640785929326,"max_completion_cost":43.27442497227064,"max_cost":89.2877376010141},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3.1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.1200000000000005e-08,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.28485877759999995,"max_cost":0.28485877759999995},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.33813975598162e-05,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":438.22202503565194,"max_cost":438.22202503565194},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5v","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":1.95597e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1330000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.04272881664,"max_completion_cost":0.03204661248,"max_cost":0.06409322496},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.0030090318491522734,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017429884328949455,"input_cache_write":0.0,"max_prompt_cost":65.7333037553478,"max_completion_cost":49.29997781651085,"max_cost":98.5999556330217},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5v","alias_ids":null,"forwarded_model_id":null},{"id":"jamba-large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5563647999999999,"max_completion_cost":0.03560734719999999,"max_cost":0.5830703103999999},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":855.9023926477577,"max_completion_cost":54.77775312945649,"max_cost":896.9857074948501},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ai21/jamba-large-1.7","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-chat","name":"OpenAI: GPT-5 Chat","created":1754587837,"description":"GPT-5 Chat is designed for advanced, natural, multimodal, and context-aware conversations for enterprise applications.","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2874999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.17386400000000002,"max_completion_cost":0.17803673600000003,"max_cost":0.32964614400000003},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0,"max_prompt_cost":267.46949770242435,"max_completion_cost":273.8887656472825,"max_cost":507.12216764379656},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-chat-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.3583125000000002e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.2874999999999998e-07,"input_cache_write":0.0,"max_prompt_cost":0.5433250000000001,"max_completion_cost":1.3909120000000001,"max_cost":1.7603730000000002},"sats_pricing":{"prompt":0.00208960545080019,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0,"max_prompt_cost":835.8421803200761,"max_completion_cost":2139.755981619395,"max_cost":2708.1286642370465},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.10866499999999998,"max_completion_cost":0.27818239999999994,"max_cost":0.3520745999999999},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":167.16843606401517,"max_completion_cost":427.95119632387883,"max_cost":541.625732847409},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.43325e-08,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.150000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.021733,"max_completion_cost":0.05563648,"max_cost":0.07041492},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":7.922674694977025e-06,"input_cache_write":0.0,"max_prompt_cost":33.43368721280304,"max_completion_cost":85.59023926477579,"max_cost":108.32514656948186},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-120b","name":"GPT-OSS 120B (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.6299749999999997e-07,"completion":6.519899999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021364408319999997,"max_completion_cost":0.08545763327999999,"max_cost":0.08545763327999999},"sats_pricing":{"prompt":0.00025075265409602276,"completion":0.001003010616384091,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.866651877673895,"max_completion_cost":131.46660751069558,"max_cost":131.46660751069558},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.1512850000000004e-08,"completion":1.52131e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0041304522752000005,"max_completion_cost":0.019940114432,"max_cost":0.019940114432},"sats_pricing":{"prompt":4.847884645856441e-05,"completion":0.00023403581048962128,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.354219363016955,"max_completion_cost":30.67554175249564,"max_cost":30.67554175249564},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-oss-20b","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.629975e-05,"completion":8.149875e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.545e-06,"input_cache_write":1.93125e-05,"max_prompt_cost":3.25995,"max_completion_cost":2.60796,"max_cost":5.346318},"sats_pricing":{"prompt":0.02507526540960228,"completion":0.12537632704801138,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0023768024084931073,"input_cache_write":0.029710030106163837,"max_prompt_cost":5015.053081920456,"max_completion_cost":4012.0424655363645,"max_cost":8224.687054349548},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-2508","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.25995e-07,"completion":9.77985e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":0.0,"max_prompt_cost":0.08345472,"max_completion_cost":0.25036416,"max_cost":0.25036416},"sats_pricing":{"prompt":0.0005015053081920456,"completion":0.0015045159245761367,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0,"max_prompt_cost":128.38535889716366,"max_completion_cost":385.156076691491,"max_cost":385.156076691491},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/codestral-2508","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":160000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.60655e-08,"completion":2.933955e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012170480000000001,"max_completion_cost":0.009613983744,"max_cost":0.01929194944},"sats_pricing":{"prompt":0.00011701790524481064,"completion":0.000451354777372841,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18.722864839169702,"max_completion_cost":14.789993344953254,"max_cost":29.678415465061004},"per_request_limits":null,"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":5.2322197500000005e-08,"completion":2.0977778250000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.006697241280000001,"max_completion_cost":0.00671288904,"max_cost":0.011735820000000001},"sats_pricing":{"prompt":8.049160196482332e-05,"completion":0.00032271866582158135,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.302925051497386,"max_completion_cost":10.326997306290604,"max_cost":18.054191094913644},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.5199e-07,"completion":2.3906300000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1330000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.08545763328,"max_completion_cost":0.23500849152000003,"max_cost":0.25637289984},"sats_pricing":{"prompt":0.0010030106163840913,"completion":0.003677705593408335,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00017429884328949455,"input_cache_write":0.0,"max_prompt_cost":131.4666075106956,"max_completion_cost":361.53317065441297,"max_cost":394.3998225320868},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5-air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.412645e-07,"completion":9.236525e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.018515820544,"max_completion_cost":0.09079873536,"max_cost":0.095427690496},"sats_pricing":{"prompt":0.00021731896688321976,"completion":0.0014209317065441293,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":28.48443162731738,"max_completion_cost":139.68327048011406,"max_cost":146.80437838694343},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5-air","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.6245417499999998e-07,"completion":1.6245417500000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021293193625599997,"max_completion_cost":0.21293193625600004,"max_cost":0.21293193625600004},"sats_pricing":{"prompt":0.0002499168119157027,"completion":0.0024991681191570275,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.75709637141498,"max_completion_cost":327.5709637141499,"max_cost":327.5709637141499},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.39063e-07,"completion":1.95597e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.062668931072,"max_completion_cost":0.12818644992,"max_cost":0.175188148224},"sats_pricing":{"prompt":0.0003677705593408334,"completion":0.0030090318491522734,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":96.40884550784344,"max_completion_cost":197.1999112660434,"max_cost":269.50654539692596},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":2.1733e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":0.0,"max_prompt_cost":0.01390912,"max_completion_cost":0.00044509184,"max_cost":0.01413166592},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0003343368721280304,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0,"max_prompt_cost":21.397559816193947,"max_completion_cost":0.6847219141182063,"max_cost":21.739920773253047},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance/ui-tars-1.5-7b","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.21e-08,"completion":2.884e-07,"request":0.0,"image":1.03e-07,"web_search":0.01442,"internal_reasoning":4.12e-07,"input_cache_read":1.0300000000000001e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.0756023296,"max_completion_cost":0.018900294,"max_cost":0.0897775501},"sats_pricing":{"prompt":0.00011091744572967834,"completion":0.00044366978291871336,"request":0.001,"image":0.00015845349389954046,"web_search":22.183489145935667,"internal_reasoning":0.0006338139755981618,"input_cache_read":1.584534938995405e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":116.3053715734432,"max_completion_cost":29.07589922357788,"max_cost":138.11229599112662},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.779850000000001e-08,"completion":5.976575000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.025637289984000004,"max_completion_cost":0.009792020480000002,"max_cost":0.03382697984000001},"sats_pricing":{"prompt":0.0001504515924576137,"completion":0.0009194263983520837,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":39.43998225320868,"max_completion_cost":15.06388211060054,"max_cost":52.03886547298369},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.193905000000001e-07,"completion":2.499295e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08118475161600001,"max_completion_cost":0.25080925184,"max_cost":0.26983692800000003},"sats_pricing":{"prompt":0.0009528600855648868,"completion":0.0038448740294723498,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":124.89327713516084,"max_completion_cost":385.8407986056093,"max_cost":415.1126604341626},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2","alias_ids":null,"forwarded_model_id":null},{"id":"dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":9.77985e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02781824,"max_completion_cost":0.00801165312,"max_cost":0.034049525760000005},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0015045159245761367,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.795119632387895,"max_completion_cost":12.324994454127712,"max_cost":52.38122643004279},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"venice/uncensored","alias_ids":null,"forwarded_model_id":null},{"id":"hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.52131e-07,"completion":6.193905000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019940114432,"max_completion_cost":0.08118475161600001,"max_cost":0.08118475161600001},"sats_pricing":{"prompt":0.00023403581048962128,"completion":0.0009528600855648868,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.67554175249564,"max_completion_cost":124.89327713516084,"max_cost":124.89327713516084},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hunyuan-a13b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-large","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.77985e-07,"completion":2.064635e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.25637289984,"max_completion_cost":0.27061583872,"max_cost":0.39880228864},"sats_pricing":{"prompt":0.0015045159245761367,"completion":0.0031762002852162886,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":394.3998225320868,"max_completion_cost":416.3109237838694,"max_cost":613.5108350499128},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"morph/morph-v3-large","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-fast","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.6932e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0712146944,"max_completion_cost":0.04955124,"max_cost":0.0877317744},"sats_pricing":{"prompt":0.0013373474885121216,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":109.555506258913,"max_completion_cost":76.22880684519093,"max_cost":134.9651085406433},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"morph/morph-v3-fast","alias_ids":null,"forwarded_model_id":null},{"id":"ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.56393e-07,"completion":1.3583125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.056136339,"max_completion_cost":0.021733000000000002,"max_cost":0.070567051},"sats_pricing":{"prompt":0.0007021074314688639,"completion":0.00208960545080019,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":86.35921407067025,"max_completion_cost":33.433687212803044,"max_cost":108.55918237997147},"per_request_limits":null,"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":8.149875e-08,"completion":2.1733e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01043184,"max_completion_cost":0.00356073472,"max_cost":0.0126572992},"sats_pricing":{"prompt":0.0001253763270480114,"completion":0.0003343368721280304,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":16.048169862145457,"max_completion_cost":5.47777531294565,"max_cost":19.47177943273649},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m1","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":2.3906300000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.43466,"max_completion_cost":0.09562520000000002,"max_cost":0.5128988000000001},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.003677705593408335,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":668.6737442560608,"max_completion_cost":147.1082237363334,"max_cost":789.0350182221518},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m1","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.163e-07,"completion":1.8025000000000001e-06,"request":0.0,"image":3.09e-07,"web_search":0.01442,"internal_reasoning":2.5750000000000003e-06,"input_cache_read":3.09e-08,"input_cache_write":8.583333333333334e-08,"max_prompt_cost":0.2268069888,"max_completion_cost":0.11812683750000001,"max_cost":0.3307586058},"sats_pricing":{"prompt":0.000332752337189035,"completion":0.0027729361432419584,"request":0.001,"image":0.0004753604816986214,"web_search":22.183489145935667,"internal_reasoning":0.0039613373474885125,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.00013204457824961708,"max_prompt_cost":348.9161147203296,"max_completion_cost":181.72437014736175,"max_cost":508.8335604500079},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.012500000000001e-07,"completion":7.2100000000000004e-06,"request":0.0,"image":1.2875000000000002e-06,"web_search":0.01442,"internal_reasoning":1.0300000000000001e-05,"input_cache_read":1.2874999999999998e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":0.9450291200000001,"max_completion_cost":0.47251456000000003,"max_cost":1.35847936},"sats_pricing":{"prompt":0.0013864680716209792,"completion":0.011091744572967834,"request":0.001,"image":0.0019806686737442562,"web_search":22.183489145935667,"internal_reasoning":0.01584534938995405,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0005942006021232768,"max_prompt_cost":1453.8171446680399,"max_completion_cost":726.9085723340199,"max_cost":2089.862145460307},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1733000000000004e-05,"completion":8.693200000000001e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.3466000000000005,"max_completion_cost":8.693200000000001,"max_cost":10.866500000000002},"sats_pricing":{"prompt":0.03343368721280304,"completion":0.13373474885121217,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6686.7374425606085,"max_completion_cost":13373.474885121217,"max_cost":16716.843606401522},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.012500000000001e-07,"completion":7.2100000000000004e-06,"request":0.0,"image":1.2875000000000002e-06,"web_search":0.01442,"internal_reasoning":1.0300000000000001e-05,"input_cache_read":1.2874999999999998e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":0.9450291200000001,"max_completion_cost":0.47251456000000003,"max_cost":1.35847936},"sats_pricing":{"prompt":0.0013864680716209792,"completion":0.011091744572967834,"request":0.001,"image":0.0019806686737442562,"web_search":22.183489145935667,"internal_reasoning":0.01584534938995405,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0005942006021232768,"max_prompt_cost":1453.8171446680399,"max_completion_cost":726.9085723340199,"max_cost":2089.862145460307},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro-preview-06-05","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":5.433249999999999e-07,"completion":2.3362975000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.605e-07,"input_cache_write":0.0,"max_prompt_cost":0.08901836799999999,"max_completion_cost":0.07655579648000001,"max_cost":0.14777049088},"sats_pricing":{"prompt":0.0008358421803200759,"completion":0.0035941213753763273,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005545872286483916,"input_cache_write":0.0,"max_prompt_cost":136.94438282364123,"max_completion_cost":117.77216922833149,"max_cost":227.32767548724448},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1-0528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.629975e-05,"completion":8.149875e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.545e-06,"input_cache_write":1.93125e-05,"max_prompt_cost":3.25995,"max_completion_cost":2.60796,"max_cost":5.346318},"sats_pricing":{"prompt":0.02507526540960228,"completion":0.12537632704801138,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0023768024084931073,"input_cache_write":0.029710030106163837,"max_prompt_cost":5015.053081920456,"max_completion_cost":4012.0424655363645,"max_cost":8224.687054349548},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4-opus-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-07,"input_cache_write":3.8625e-06,"max_prompt_cost":3.2599500000000003,"max_completion_cost":1.043184,"max_cost":4.0944972},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0004753604816986214,"input_cache_write":0.005942006021232768,"max_prompt_cost":5015.053081920457,"max_completion_cost":1604.8169862145458,"max_cost":6298.906670892093},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4-sonnet-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.519899999999999e-08,"completion":1.3039799999999997e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0021364408319999996,"max_completion_cost":0.004272881663999999,"max_cost":0.004272881663999999},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0002006021232768182,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.2866651877673894,"max_completion_cost":6.573330375534779,"max_cost":6.573330375534779},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3n-e4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.1200000000000005e-08,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.28485877759999995,"max_cost":0.28485877759999995},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.33813975598162e-05,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":438.22202503565194,"max_cost":438.22202503565194},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.012500000000001e-07,"completion":7.2100000000000004e-06,"request":0.0,"image":1.2875000000000002e-06,"web_search":0.01442,"internal_reasoning":1.0300000000000001e-05,"input_cache_read":1.2874999999999998e-07,"input_cache_write":3.8625e-07,"max_prompt_cost":0.9450291200000001,"max_completion_cost":0.47250735000000005,"max_cost":1.35847305125},"sats_pricing":{"prompt":0.0013864680716209792,"completion":0.011091744572967834,"request":0.001,"image":0.0019806686737442562,"web_search":22.183489145935667,"internal_reasoning":0.01584534938995405,"input_cache_read":0.00019806686737442556,"input_cache_write":0.0005942006021232768,"max_prompt_cost":1453.8171446680399,"max_completion_cost":726.897480589447,"max_cost":2089.852440183806},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro-preview-03-25","alias_ids":null,"forwarded_model_id":null},{"id":"virtuoso-large","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.149875000000001e-07,"completion":1.30398e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.10682204160000001,"max_completion_cost":0.08345472,"max_cost":0.1381175616},"sats_pricing":{"prompt":0.0012537632704801142,"completion":0.0020060212327681825,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":164.33325938836953,"max_completion_cost":128.38535889716366,"max_cost":212.47776897480585},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"arcee-ai/virtuoso-large","alias_ids":null,"forwarded_model_id":null},{"id":"coder-large","name":"Arcee AI: Coder Large","created":1746478663,"description":"Coder‑Large is a 32 B‑parameter offspring of Qwen 2.5‑Instruct that has been further trained on permissively‑licensed GitHub, CodeSearchNet and synthetic bug‑fix corpora. It supports a 32k context window, enabling multi‑file...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.433249999999999e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017803673599999997,"max_completion_cost":0.02848587776,"max_cost":0.02848587776},"sats_pricing":{"prompt":0.0008358421803200759,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.388876564728246,"max_completion_cost":43.8222025035652,"max_cost":43.8222025035652},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"arcee-ai/coder-large","alias_ids":null,"forwarded_model_id":null},{"id":"llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":163840,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.9559700000000003e-07,"completion":1.9559700000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.03204661248000001,"max_completion_cost":0.0032046612480000005,"max_cost":0.03204661248000001},"sats_pricing":{"prompt":0.0003009031849152274,"completion":0.0003009031849152274,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":49.29997781651086,"max_completion_cost":4.929997781651085,"max_cost":49.29997781651086},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-guard-4-12b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.3039799999999997e-07,"completion":5.433249999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005341102079999999,"max_completion_cost":0.008901836799999998,"max_cost":0.012106498047999997},"sats_pricing":{"prompt":0.0002006021232768182,"completion":0.0008358421803200759,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.216662969418474,"max_completion_cost":13.694438282364123,"max_cost":18.624436064015203},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-8b","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.2713805e-07,"completion":4.9442575e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0166642384896,"max_completion_cost":0.004050335744,"max_cost":0.019673059328000002},"sats_pricing":{"prompt":0.0001955870701948978,"completion":0.0007606163840912692,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":25.635988464585644,"max_completion_cost":6.230969418475677,"max_cost":30.26470860402472},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-8b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-14b","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131702,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.08665e-07,"completion":2.6079599999999995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0044509184,"max_completion_cost":0.010682204159999998,"max_cost":0.010682204159999998},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0004012042465536364,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.847219141182062,"max_completion_cost":16.433325938836948,"max_cost":16.433325938836948},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":40960,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-14b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-32b","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":8.6932e-08,"completion":3.04262e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00356073472,"max_completion_cost":0.004985028608,"max_cost":0.00712146944},"sats_pricing":{"prompt":0.00013373474885121215,"completion":0.00046807162097924255,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.47777531294565,"max_completion_cost":7.66888543812391,"max_cost":10.9555506258913},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-32b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":4.9442575e-07,"completion":1.977703e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.064805371904,"max_completion_cost":0.016201342976,"max_cost":0.076956379136},"sats_pricing":{"prompt":0.0007606163840912692,"completion":0.0030424655363650768,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":99.69551069561084,"max_completion_cost":24.92387767390271,"max_cost":118.38841895103785},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1953150000000002e-06,"completion":4.781260000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.8325e-07,"input_cache_write":0.0,"max_prompt_cost":0.23906300000000005,"max_completion_cost":0.4781260000000001,"max_cost":0.5976575000000002},"sats_pricing":{"prompt":0.0018388527967041675,"completion":0.00735541118681667,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0004357471082237363,"input_cache_write":0.0,"max_prompt_cost":367.77055934083353,"max_completion_cost":735.5411186816671,"max_cost":919.4263983520839},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":0.43465999999999994,"max_completion_cost":0.8693199999999999,"max_cost":1.08665},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":668.6737442560607,"max_completion_cost":1337.3474885121213,"max_cost":1671.684360640152},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1953150000000002e-06,"completion":4.781260000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.8325e-07,"input_cache_write":0.0,"max_prompt_cost":0.23906300000000005,"max_completion_cost":0.4781260000000001,"max_cost":0.5976575000000002},"sats_pricing":{"prompt":0.0018388527967041675,"completion":0.00735541118681667,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0004357471082237363,"input_cache_write":0.0,"max_prompt_cost":367.77055934083353,"max_completion_cost":735.5411186816671,"max_cost":919.4263983520839},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.149999999999999e-07,"input_cache_write":0.0,"max_prompt_cost":2.2766969207999996,"max_completion_cost":9.106787683199999,"max_cost":9.106787683199999},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0007922674694977023,"input_cache_write":0.0,"max_prompt_cost":3502.432831563935,"max_completion_cost":14009.73132625574,"max_cost":14009.73132625574},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":1.73864e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":1.03e-07,"input_cache_write":0.0,"max_prompt_cost":0.45533938416,"max_completion_cost":0.05697175552,"max_cost":0.4980682008},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0026746949770242432,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.00015845349389954046,"input_cache_write":0.0,"max_prompt_cost":700.4865663127872,"max_completion_cost":87.6444050071304,"max_cost":766.2198700681349},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":2.575e-08,"input_cache_write":0.0,"max_prompt_cost":0.11383484604,"max_completion_cost":0.01424293888,"max_cost":0.1245170502},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":3.9613373474885114e-05,"input_cache_write":0.0,"max_prompt_cost":175.1216415781968,"max_completion_cost":21.9111012517826,"max_cost":191.55496751703373},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-maverick","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.22788702208,"max_completion_cost":0.01424293888,"max_cost":0.23856922624},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":350.5776200285216,"max_completion_cost":21.9111012517826,"max_cost":367.0109459673585},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-scout","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":10000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":3.25995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0356073472,"max_completion_cost":0.00534110208,"max_cost":0.03916808192},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0005015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":54.7777531294565,"max_completion_cost":8.216662969418476,"max_cost":60.25552844240215},"per_request_limits":null,"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.6079599999999995e-07,"completion":9.77985e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3905000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.04272881663999999,"max_completion_cost":0.01602330624,"max_cost":0.05447924121599999},"sats_pricing":{"prompt":0.0004012042465536364,"completion":0.0015045159245761367,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021391221676437966,"input_cache_write":0.0,"max_prompt_cost":65.73330375534779,"max_completion_cost":24.649988908255423,"max_cost":83.80996228806843},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3-0324","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":0.0001629975,"completion":0.00065199,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.5995,"max_completion_cost":65.199,"max_cost":81.49875},"sats_pricing":{"prompt":0.25075265409602276,"completion":1.003010616384091,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":50150.530819204556,"max_completion_cost":100301.06163840911,"max_cost":125376.3270480114},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.8141415e-07,"completion":6.0309075e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0488210112,"max_completion_cost":0.077195616,"max_cost":0.077195616},"sats_pricing":{"prompt":0.0005867612105846933,"completion":0.0009277848201552844,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":75.10543495484075,"max_completion_cost":118.75645697987639,"max_cost":118.75645697987639},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.43325e-08,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.00178036736,"max_cost":0.00801165312},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":2.738887656472825,"max_cost":12.324994454127712},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.43325e-08,"completion":1.629975e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.00267055104,"max_cost":0.0089018368},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0002507526540960228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":4.108331484709238,"max_cost":13.694438282364125},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-12b-it","alias_ids":null,"forwarded_model_id":null},{"id":"command-a","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6954560000000001,"max_completion_cost":0.08901836800000001,"max_cost":0.762219776},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1069.8779908096974,"max_completion_cost":136.94438282364126,"max_cost":1172.5862779274282},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-a-03-2025","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-search-preview","name":"OpenAI: GPT-4o-mini Search Preview","created":1741818122,"description":"GPT-4o mini Search Preview is a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.028325,"internal_reasoning":0.0,"input_cache_read":7.725e-08,"input_cache_write":0.0,"max_prompt_cost":0.02086368,"max_completion_cost":0.01068220416,"max_cost":0.02887533312},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":43.57471082237363,"internal_reasoning":0.0,"input_cache_read":0.00011884012042465535,"input_cache_write":0.0,"max_prompt_cost":32.096339724290914,"max_completion_cost":16.43332593883695,"max_cost":44.421334178418626},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-mini-search-preview-2025-03-11","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-search-preview","name":"OpenAI: GPT-4o Search Preview","created":1741817949,"description":"GPT-4o Search Previewis a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.036050000000000006,"internal_reasoning":0.0,"input_cache_read":1.2875000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.17803673600000003,"max_cost":0.4812555520000001},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":55.458722864839174,"internal_reasoning":0.0,"input_cache_read":0.0019806686737442562,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":273.8887656472825,"max_cost":740.3555696403106},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-search-preview-2025-03-11","alias_ids":null,"forwarded_model_id":null},{"id":"reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.08665e-07,"completion":2.1733e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.01424293888,"max_cost":0.01424293888},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0003343368721280304,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":21.9111012517826,"max_cost":21.9111012517826},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"rekaai/reka-flash-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":8.6932e-08,"completion":1.73864e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.011394351104,"max_completion_cost":0.002848587776,"max_cost":0.012818644992},"sats_pricing":{"prompt":0.00013373474885121215,"completion":0.0002674694977024243,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17.52888100142608,"max_completion_cost":4.38222025035652,"max_cost":19.71999112660434},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.976575000000001e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5749999999999997e-07,"input_cache_write":0.0,"max_prompt_cost":0.019584040960000004,"max_completion_cost":0.02848587776,"max_cost":0.02848587776},"sats_pricing":{"prompt":0.0009194263983520837,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003961337347488511,"input_cache_write":0.0,"max_prompt_cost":30.12776422120108,"max_completion_cost":43.8222025035652,"max_cost":43.8222025035652},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/skyfall-36b-v2","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27818239999999994,"max_completion_cost":1.1127295999999998,"max_cost":1.1127295999999998},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":427.95119632387883,"max_completion_cost":1711.8047852955153,"max_cost":1711.8047852955153},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-reasoning-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6519900000000001,"max_completion_cost":0.130398,"max_cost":0.7563084000000001},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1003.0106163840912,"max_completion_cost":200.60212327681822,"max_cost":1163.4923150055458},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-deep-research","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.1732999999999996e-06,"completion":8.693199999999998e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":3.09e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27818239999999994,"max_completion_cost":1.1127295999999998,"max_cost":1.1127295999999998},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.013373474885121214,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.004753604816986215,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":427.95119632387883,"max_completion_cost":1711.8047852955153,"max_cost":1711.8047852955153},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-deep-research","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-saba","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.0600000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.02136440832,"max_cost":0.02136440832},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.16906987799081e-05,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":32.8666518776739,"max_cost":32.8666518776739},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-saba-2502","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1953150000000002e-06,"completion":4.781260000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.665e-07,"input_cache_write":0.0,"max_prompt_cost":0.23906300000000005,"max_completion_cost":0.4781260000000001,"max_cost":0.5976575000000002},"sats_pricing":{"prompt":0.0018388527967041675,"completion":0.00735541118681667,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0008714942164474726,"input_cache_write":0.0,"max_prompt_cost":367.77055934083353,"max_completion_cost":735.5411186816671,"max_cost":919.4263983520839},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.6932e-07,"completion":1.73864e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02848587776,"max_completion_cost":0.05697175552,"max_cost":0.05697175552},"sats_pricing":{"prompt":0.0013373474885121216,"completion":0.0026746949770242432,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":43.8222025035652,"max_completion_cost":87.6444050071304,"max_cost":87.6444050071304},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":8.149875000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.008693199999999998,"max_completion_cost":0.026079600000000005,"max_cost":0.026079600000000005},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0012537632704801142,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.373474885121214,"max_completion_cost":40.12042465536366,"max_cost":40.12042465536366},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.82529e-07,"completion":8.47587e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.3560000000000004e-08,"input_cache_write":3.3475e-07,"max_prompt_cost":0.282529,"max_completion_cost":0.027773730816,"max_cost":0.301044820544},"sats_pricing":{"prompt":0.00043463793376643953,"completion":0.0013039138012993186,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.239581682776106e-05,"input_cache_write":0.0005149738551735065,"max_prompt_cost":434.63793376643946,"max_completion_cost":42.72664744097607,"max_cost":463.1223653937569},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-01-25","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1953150000000002e-06,"completion":4.781260000000001e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":5.665e-07,"input_cache_write":0.0,"max_prompt_cost":0.23906300000000005,"max_completion_cost":0.4781260000000001,"max_cost":0.5976575000000002},"sats_pricing":{"prompt":0.0018388527967041675,"completion":0.00735541118681667,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.0008714942164474726,"input_cache_write":0.0,"max_prompt_cost":367.77055934083353,"max_completion_cost":735.5411186816671,"max_cost":919.4263983520839},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.43325e-08,"completion":8.6932e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00178036736,"max_completion_cost":0.001424293888,"max_cost":0.002314477568},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.00013373474885121215,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.738887656472825,"max_completion_cost":2.19111012517826,"max_cost":3.5605539534146726},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","alias_ids":null,"forwarded_model_id":null},{"id":"sonar","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.00515,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.13808278879999997,"max_completion_cost":0.13808278879999997,"max_cost":0.13808278879999997},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":7.922674694977023,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":212.42427507526534,"max_completion_cost":212.42427507526534,"max_cost":212.42427507526534},"per_request_limits":null,"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":8.6932e-07,"completion":8.6932e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.00712146944,"max_cost":0.00712146944},"sats_pricing":{"prompt":0.0013373474885121216,"completion":0.0013373474885121216,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":10.9555506258913,"max_cost":10.9555506258913},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":7.60655e-07,"completion":2.7166250000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.048681920000000004,"max_completion_cost":0.043466000000000005,"max_cost":0.07997744000000001},"sats_pricing":{"prompt":0.0011701790524481063,"completion":0.00417921090160038,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":74.89145935667881,"max_completion_cost":66.86737442560609,"max_cost":123.03596894311521},"per_request_limits":null,"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.1733e-07,"completion":1.1953150000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.21737172736,"max_completion_cost":1.1955445004800003,"max_cost":1.1955445004800003},"sats_pricing":{"prompt":0.0003343368721280304,"completion":0.0018388527967041675,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":334.401064807479,"max_completion_cost":1839.2058564411348,"max_cost":1839.2058564411348},"per_request_limits":null,"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-01","alias_ids":null,"forwarded_model_id":null},{"id":"phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.60655e-08,"completion":1.52131e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001246257152,"max_completion_cost":0.002492514304,"max_cost":0.002492514304},"sats_pricing":{"prompt":0.00011701790524481064,"completion":0.00023403581048962128,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9172213595309775,"max_completion_cost":3.834442719061955,"max_cost":3.834442719061955},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"microsoft/phi-4","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-70b-hanami-x1","name":"Sao10K: Llama 3.1 70B Hanami x1","created":1736302854,"description":"This is [Sao10K](/sao10k)'s experiment over [Euryale v2.2](/sao10k/l3.1-euryale-70b).","context_length":16000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":3.2599500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05215920000000001,"max_completion_cost":0.05215920000000001,"max_cost":0.05215920000000001},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.005015053081920457,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":80.24084931072731,"max_completion_cost":80.24084931072731,"max_cost":80.24084931072731},"per_request_limits":null,"top_provider":{"context_length":16000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3.1-70b-hanami-x1","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.1754733000000001e-07,"completion":8.69428665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.884e-08,"input_cache_write":0.0,"max_prompt_cost":0.027846058240000002,"max_completion_cost":0.01391085864,"max_cost":0.038276159600000005},"sats_pricing":{"prompt":0.00033467120900015847,"completion":0.0013375146569481857,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.436697829187133e-05,"input_cache_write":0.0,"max_prompt_cost":42.83791475202028,"max_completion_cost":21.40023451117097,"max_cost":58.88340991918872},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3","alias_ids":null,"forwarded_model_id":null},{"id":"l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":7.063225e-07,"completion":8.149875000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09257910272,"max_completion_cost":0.013352755200000002,"max_cost":0.09435947008},"sats_pricing":{"prompt":0.0010865948344160987,"completion":0.0012537632704801142,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":142.4221581365869,"max_completion_cost":20.54165742354619,"max_cost":145.1610457930597},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","alias_ids":null,"forwarded_model_id":null},{"id":"o1","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.629975e-05,"completion":6.5199e-05,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":7.725e-06,"input_cache_write":0.0,"max_prompt_cost":3.25995,"max_completion_cost":6.5199,"max_cost":8.149875},"sats_pricing":{"prompt":0.02507526540960228,"completion":0.10030106163840911,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":0.011884012042465536,"input_cache_write":0.0,"max_prompt_cost":5015.053081920456,"max_completion_cost":10030.106163840912,"max_cost":12537.632704801139},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":4.0749375e-08,"completion":1.629975e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00521592,"max_completion_cost":0.00065199,"max_cost":0.0057049125},"sats_pricing":{"prompt":6.26881635240057e-05,"completion":0.0002507526540960228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.024084931072728,"max_completion_cost":1.003010616384091,"max_cost":8.776342893360798},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r7b-12-2024","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":1.08665e-07,"completion":3.47728e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01424293888,"max_completion_cost":0.005697175552,"max_cost":0.018159747072},"sats_pricing":{"prompt":0.0001671684360640152,"completion":0.0005349389954048486,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.9111012517826,"max_completion_cost":8.76444050071304,"max_cost":27.936654096022814},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.3-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"nova-lite-v1","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":6.519899999999999e-08,"completion":2.6079599999999995e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019559699999999996,"max_completion_cost":0.0013352755199999998,"max_cost":0.020561156639999995},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0004012042465536364,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.09031849152273,"max_completion_cost":2.0541657423546185,"max_cost":31.63094279828869},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-micro-v1","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.803275e-08,"completion":1.52131e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004868192,"max_completion_cost":0.00077891072,"max_cost":0.0054523750400000005},"sats_pricing":{"prompt":5.850895262240532e-05,"completion":0.00023403581048962128,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.489145935667882,"max_completion_cost":1.1982633497068609,"max_cost":8.387843447948027},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-micro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-pro-v1","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":8.6932e-07,"completion":3.47728e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.26079600000000003,"max_completion_cost":0.0178036736,"max_cost":0.27414875520000004},"sats_pricing":{"prompt":0.0013373474885121216,"completion":0.0053493899540484864,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":401.2042465536365,"max_completion_cost":27.38887656472825,"max_cost":421.7459039771827},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-pro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2875000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.17803673600000003,"max_cost":0.4812555520000001},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0019806686737442562,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":273.8887656472825,"max_cost":740.3555696403106},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-11-20","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2407","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":0.28485877759999995,"max_completion_cost":0.8545763328000001,"max_cost":0.8545763328000001},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":438.22202503565194,"max_completion_cost":1314.6660751069562,"max_cost":1314.6660751069562},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large-2407","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":7.17189e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.023500849152,"max_completion_cost":0.03560734719999999,"max_cost":0.03560734719999999},"sats_pricing":{"prompt":0.0011033116780225004,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":36.15331706544129,"max_completion_cost":54.77775312945649,"max_cost":54.77775312945649},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":4.3466e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01424293888,"max_completion_cost":0.01424293888,"max_cost":0.01424293888},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.9111012517826,"max_completion_cost":21.9111012517826,"max_cost":21.9111012517826},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/unslopnemo-12b","alias_ids":null,"forwarded_model_id":null},{"id":"magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.2599500000000004e-06,"completion":5.433250000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05341102080000001,"max_completion_cost":0.011127296000000002,"max_cost":0.05786193920000001},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.00835842180320076,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":82.16662969418476,"max_completion_cost":17.118047852955158,"max_cost":89.01384883536683},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthracite-org/magnum-v4-72b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":4.3466e-08,"completion":1.08665e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.001424293888,"max_completion_cost":0.00356073472,"max_cost":0.00356073472},"sats_pricing":{"prompt":6.686737442560608e-05,"completion":0.0001671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.19111012517826,"max_completion_cost":5.47777531294565,"max_cost":5.47777531294565},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-7b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"inflection-3-productivity","name":"Inflection: Inflection 3 Productivity","created":1728604800,"description":"Inflection 3 Productivity is optimized for following instructions. It is better for tasks requiring JSON output or precise adherence to provided guidelines. It has access to recent news. For emotional...","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021733000000000002,"max_completion_cost":0.011127296000000002,"max_cost":0.030078472000000005},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":33.433687212803044,"max_completion_cost":17.118047852955158,"max_cost":46.27222310251941},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":1024,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inflection/inflection-3-productivity","alias_ids":null,"forwarded_model_id":null},{"id":"inflection-3-pi","name":"Inflection: Inflection 3 Pi","created":1728604800,"description":"Inflection 3 Pi powers Inflection's [Pi](https://pi.ai) chatbot, including backstory, emotional intelligence, productivity, and safety. It has access to recent news, and excels in scenarios like customer support and roleplay. Pi...","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021733000000000002,"max_completion_cost":0.011127296000000002,"max_cost":0.030078472000000005},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":33.433687212803044,"max_completion_cost":17.118047852955158,"max_cost":46.27222310251941},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":1024,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inflection/inflection-3-pi","alias_ids":null,"forwarded_model_id":null},{"id":"rocinante-12b","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":2.7166249999999995e-07,"completion":5.433249999999999e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017803673599999997,"max_completion_cost":0.03560734719999999,"max_cost":0.03560734719999999},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.0008358421803200759,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.388876564728246,"max_completion_cost":54.77775312945649,"max_cost":54.77775312945649},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/rocinante-12b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.9339550000000002e-08,"completion":2.1841665000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0017603730000000002,"max_completion_cost":0.013104999,"max_cost":0.013104999},"sats_pricing":{"prompt":4.513547773728411e-05,"completion":0.0003360085564886706,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.7081286642370466,"max_completion_cost":20.160513389320233,"max_cost":20.160513389320233},"per_request_limits":null,"top_provider":{"context_length":60000,"max_completion_tokens":60000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.2-1b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-11b-vision-instruct","name":"Meta: Llama 3.2 11B Vision Instruct","created":1727222400,"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":3.7489425e-07,"completion":3.7489425e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.049138139136,"max_completion_cost":0.006142267392,"max_cost":0.049138139136},"sats_pricing":{"prompt":0.0005767311044208524,"completion":0.0005767311044208524,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":75.59329931864997,"max_completion_cost":9.449162414831246,"max_cost":75.59329931864997},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.2-11b-vision-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.43325e-08,"completion":3.585945e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00712146944,"max_completion_cost":0.047001698304,"max_cost":0.047001698304},"sats_pricing":{"prompt":8.35842180320076e-05,"completion":0.0005516558390112502,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.9555506258913,"max_completion_cost":72.30663413088259,"max_cost":72.30663413088259},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.2-3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.9119400000000006e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012818644992000002,"max_completion_cost":0.00712146944,"max_cost":0.013530791936},"sats_pricing":{"prompt":0.0006018063698304548,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.71999112660434,"max_completion_cost":10.9555506258913,"max_cost":20.81554618919347},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-08-2024","name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02086368,"max_completion_cost":0.00260796,"max_cost":0.02281965},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.096339724290914,"max_completion_cost":4.012042465536364,"max_cost":35.10537157344319},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.043466000000000005,"max_cost":0.38032750000000004},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":66.86737442560609,"max_cost":585.0895262240532},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r-plus-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":9.236525e-07,"completion":9.236525e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.12106498048,"max_completion_cost":0.01513312256,"max_cost":0.12106498048},"sats_pricing":{"prompt":0.0014209317065441293,"completion":0.0014209317065441293,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":186.2443606401521,"max_completion_cost":23.280545080019014,"max_cost":186.2443606401521},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3.1-euryale-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":7.60655e-07,"completion":7.60655e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09970057216,"max_completion_cost":0.01246257152,"max_cost":0.09970057216},"sats_pricing":{"prompt":0.0011701790524481063,"completion":0.0011701790524481063,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":153.3777087624782,"max_completion_cost":19.172213595309774,"max_cost":153.3777087624782},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":1.0866499999999998e-06,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14242938879999997,"max_completion_cost":0.017803673599999997,"max_cost":0.14242938879999997},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":219.11101251782597,"max_completion_cost":27.388876564728246,"max_cost":219.11101251782597},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","alias_ids":null,"forwarded_model_id":null},{"id":"l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.3466e-08,"completion":5.43325e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.000356073472,"max_completion_cost":0.00044509184,"max_cost":0.00044509184},"sats_pricing":{"prompt":6.686737442560608e-05,"completion":8.35842180320076e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.547777531294565,"max_completion_cost":0.6847219141182063,"max_cost":0.6847219141182063},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3-lunaris-8b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2875000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.17803673600000003,"max_cost":0.4812555520000001},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0019806686737442562,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":273.8887656472825,"max_cost":740.3555696403106},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-08-06","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":2.1733e-08,"completion":3.2599499999999994e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002848587776,"max_completion_cost":0.0005341102079999999,"max_cost":0.003026624512},"sats_pricing":{"prompt":3.343368721280304e-05,"completion":5.015053081920455e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.38222025035652,"max_completion_cost":0.8216662969418473,"max_cost":4.656109016003803},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.1-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.3466e-07,"completion":4.3466e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05697175552,"max_completion_cost":0.00712146944,"max_cost":0.05697175552},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0006686737442560608,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":87.6444050071304,"max_completion_cost":10.9555506258913,"max_cost":87.6444050071304},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.1-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-nemo","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.1733e-08,"completion":3.2599499999999994e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002848587776,"max_completion_cost":0.004272881663999999,"max_cost":0.004272881663999999},"sats_pricing":{"prompt":3.343368721280304e-05,"completion":5.015053081920455e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.38222025035652,"max_completion_cost":6.573330375534779,"max_cost":6.573330375534779},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-nemo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.725e-08,"input_cache_write":0.0,"max_prompt_cost":0.02086368,"max_completion_cost":0.01068220416,"max_cost":0.02887533312},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011884012042465535,"input_cache_write":0.0,"max_prompt_cost":32.096339724290914,"max_completion_cost":16.43332593883695,"max_cost":44.421334178418626},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-mini-2024-07-18","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini","name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.629975e-07,"completion":6.5199e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.725e-08,"input_cache_write":0.0,"max_prompt_cost":0.02086368,"max_completion_cost":0.01068220416,"max_cost":0.02887533312},"sats_pricing":{"prompt":0.0002507526540960228,"completion":0.0010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00011884012042465535,"input_cache_write":0.0,"max_prompt_cost":32.096339724290914,"max_completion_cost":16.43332593883695,"max_cost":44.421334178418626},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":7.063225e-07,"completion":7.063225e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00578619392,"max_completion_cost":0.00144654848,"max_cost":0.00578619392},"sats_pricing":{"prompt":0.0010865948344160987,"completion":0.0010865948344160987,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.90138488353668,"max_completion_cost":2.22534622088417,"max_cost":8.90138488353668},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-2-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433250000000001e-06,"completion":1.629975e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6954560000000001,"max_completion_cost":0.066763776,"max_cost":0.7399651840000001},"sats_pricing":{"prompt":0.00835842180320076,"completion":0.02507526540960228,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1069.8779908096974,"max_completion_cost":102.70828711773093,"max_cost":1138.350182221518},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-05-13","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o","name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.7166250000000004e-06,"completion":1.0866500000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.2875000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.34772800000000004,"max_completion_cost":0.17803673600000003,"max_cost":0.4812555520000001},"sats_pricing":{"prompt":0.00417921090160038,"completion":0.01671684360640152,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0019806686737442562,"input_cache_write":0.0,"max_prompt_cost":534.9389954048487,"max_completion_cost":273.8887656472825,"max_cost":740.3555696403106},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.1732999999999996e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":0.14242938879999997,"max_completion_cost":0.42728816640000006,"max_cost":0.42728816640000006},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":219.11101251782597,"max_completion_cost":657.3330375534781,"max_cost":657.3330375534781},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mixtral-8x22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"wizardlm-2-8x22b","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":6.73723e-07,"completion":6.73723e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.044152436805,"max_completion_cost":0.005389784,"max_cost":0.044152436805},"sats_pricing":{"prompt":0.0010364443035968942,"completion":0.0010364443035968942,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":67.92337743622247,"max_completion_cost":8.291554428775154,"max_cost":67.92337743622247},"per_request_limits":null,"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"microsoft/wizardlm-2-8x22b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3909120000000001,"max_completion_cost":0.133527552,"max_cost":1.4799303680000002},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2139.755981619395,"max_completion_cost":205.41657423546187,"max_cost":2276.700364443036},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"claude-3-haiku","name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.7166249999999995e-07,"completion":1.3583125000000002e-06,"request":0.0,"image":0.0,"web_search":0.0103,"internal_reasoning":0.0,"input_cache_read":3.09e-08,"input_cache_write":3.09e-07,"max_prompt_cost":0.05433249999999999,"max_completion_cost":0.005563648000000001,"max_cost":0.058783418399999995},"sats_pricing":{"prompt":0.00041792109016003794,"completion":0.00208960545080019,"request":0.001,"image":0.0,"web_search":15.845349389954047,"internal_reasoning":0.0,"input_cache_read":4.753604816986214e-05,"input_cache_write":0.0004753604816986214,"max_prompt_cost":83.58421803200758,"max_completion_cost":8.559023926477579,"max_cost":90.43143717318965},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-3-haiku","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large","name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.1732999999999996e-06,"completion":6.519900000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.06e-07,"input_cache_write":0.0,"max_prompt_cost":0.27818239999999994,"max_completion_cost":0.8345472000000002,"max_cost":0.8345472000000002},"sats_pricing":{"prompt":0.0033433687212803035,"completion":0.010030106163840913,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003169069877990809,"input_cache_write":0.0,"max_prompt_cost":427.95119632387883,"max_completion_cost":1283.853588971637,"max_cost":1283.853588971637},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866499999999998e-06,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004449831749999999,"max_completion_cost":0.008899663499999998,"max_cost":0.008899663499999998},"sats_pricing":{"prompt":0.0016716843606401517,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.845547456821421,"max_completion_cost":13.691094913642843,"max_cost":13.691094913642843},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-0613","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.0866500000000002e-05,"completion":3.25995e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.3909120000000001,"max_completion_cost":0.133527552,"max_cost":1.4799303680000002},"sats_pricing":{"prompt":0.01671684360640152,"completion":0.05015053081920456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2139.755981619395,"max_completion_cost":205.41657423546187,"max_cost":2276.700364443036},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4-turbo-preview","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":1.6299750000000002e-06,"completion":2.1732999999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0066747476250000005,"max_completion_cost":0.008899663499999998,"max_cost":0.008899663499999998},"sats_pricing":{"prompt":0.0025075265409602284,"completion":0.0033433687212803035,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.268321185232134,"max_completion_cost":13.691094913642843,"max_cost":13.691094913642843},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.2599500000000004e-06,"completion":4.346599999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05341428075000001,"max_completion_cost":0.017803673599999997,"max_cost":0.057865199150000005},"sats_pricing":{"prompt":0.005015053081920457,"completion":0.006686737442560607,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":82.17164474726668,"max_completion_cost":27.388876564728246,"max_cost":89.01886388844873},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-16k","alias_ids":null,"forwarded_model_id":null},{"id":"weaver","name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":8.149875000000001e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.006519900000000001,"max_completion_cost":0.0021732999999999995,"max_cost":0.007063225},"sats_pricing":{"prompt":0.0012537632704801142,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.030106163840914,"max_completion_cost":3.3433687212803034,"max_cost":10.865948344160989},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":2000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mancer/weaver","alias_ids":null,"forwarded_model_id":null},{"id":"remm-slerp-l2-13b","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":4.889925e-07,"completion":7.063225e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00300436992,"max_completion_cost":0.00289309696,"max_cost":0.0038945536},"sats_pricing":{"prompt":0.0007522579622880683,"completion":0.0010865948344160987,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.621872920297892,"max_completion_cost":4.45069244176834,"max_cost":5.991316748534305},"per_request_limits":null,"top_provider":{"context_length":6144,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"undi95/remm-slerp-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"mythomax-l2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":6.519899999999999e-08,"completion":6.519899999999999e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00026705510399999995,"max_completion_cost":0.00026705510399999995,"max_cost":0.00026705510399999995},"sats_pricing":{"prompt":0.0001003010616384091,"completion":0.0001003010616384091,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.41083314847092367,"max_completion_cost":0.41083314847092367,"max_cost":0.41083314847092367},"per_request_limits":null,"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"gryphe/mythomax-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4","name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.25995e-05,"completion":6.5199e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2670225045,"max_completion_cost":0.267055104,"max_cost":0.4005500565},"sats_pricing":{"prompt":0.05015053081920456,"completion":0.10030106163840911,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":410.7829979401045,"max_completion_cost":410.83314847092373,"max_cost":616.1995721755663},"per_request_limits":null,"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo","name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.433249999999999e-07,"completion":1.6299750000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.008902380124999998,"max_completion_cost":0.006676377600000001,"max_cost":0.013353298525},"sats_pricing":{"prompt":0.0008358421803200759,"completion":0.0025075265409602284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":13.695274124544442,"max_completion_cost":10.270828711773095,"max_cost":20.542493265726506},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"autoclaw","name":"🔥 AutoClaw (for OpenClaw)","created":1783935094,"description":"PPQ.AI model","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":-1.0299999999999999e-06,"completion":-1.0299999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-0.206,"max_completion_cost":-0.206,"max_cost":-0.206},"sats_pricing":{"prompt":-0.0015845349389954045,"completion":-0.0015845349389954045,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-316.9069877990809,"max_completion_cost":-316.9069877990809,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"llama3-3-70b","name":"Llama 3.3 70B (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.9016375e-06,"completion":2.9882875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2492514304,"max_completion_cost":0.3916808192,"max_cost":0.3916808192},"sats_pricing":{"prompt":0.002925447631120266,"completion":0.004597131991760418,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":383.4442719061955,"max_completion_cost":602.5552844240215,"max_cost":602.5552844240215},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b","name":"Qwen3-VL 30B (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.3583124999999998e-06,"completion":4.346599999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.35607347199999995,"max_completion_cost":1.1394351103999998,"max_cost":1.1394351103999998},"sats_pricing":{"prompt":0.0020896054508001897,"completion":0.006686737442560607,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":547.7775312945649,"max_completion_cost":1752.8881001426078,"max_cost":1752.8881001426078},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-2","name":"GLM-5.2 (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":384000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.6299750000000002e-06,"completion":5.704912499999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6259104000000001,"max_completion_cost":2.1906863999999997,"max_cost":2.1906863999999997},"sats_pricing":{"prompt":0.0025075265409602284,"completion":0.008776342893360796,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":962.8901917287277,"max_completion_cost":3370.115671050546,"max_cost":3370.115671050546},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"gemma4-31b","name":"Gemma 4 31B (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":4.3466e-07,"completion":1.0866499999999998e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11394351104,"max_completion_cost":0.28485877759999995,"max_cost":0.28485877759999995},"sats_pricing":{"prompt":0.0006686737442560608,"completion":0.0016716843606401517,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":175.2888100142608,"max_completion_cost":438.22202503565194,"max_cost":438.22202503565194},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-6","name":"Kimi K2.6 (Private via TEE)","created":1783935094,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.6299750000000002e-06,"completion":5.704912499999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.42728816640000006,"max_completion_cost":1.4955085823999998,"max_cost":1.4955085823999998},"sats_pricing":{"prompt":0.0025075265409602284,"completion":0.008776342893360796,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":657.3330375534781,"max_completion_cost":2300.6656314371726,"max_cost":2300.6656314371726},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2-fast","name":"GLM 5.2 (Fast)","created":1783935094,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":2.2819650000000004e-06,"completion":7.17189e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.3928137318400005,"max_completion_cost":7.52027172864,"max_cost":7.52027172864},"sats_pricing":{"prompt":0.0035105371573443196,"completion":0.011033116780225003,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3681.0650102994773,"max_completion_cost":11569.061460941213,"max_cost":11569.061460941213},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"sakana-fugu-ultra","name":"fugu-ultra","created":1784088297,"description":"fugu-ultra via sakana-ai-team","context_length":1000000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":5.1509999999999995e-06,"completion":3.0401e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.151,"max_completion_cost":30.401,"max_cost":30.401},"sats_pricing":{"prompt":0.00792421307841294,"completion":0.04676839483533912,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7924.213078412941,"max_completion_cost":46768.394835339124,"max_cost":46768.394835339124},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"generic","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"sakana-fugu-ultra"},{"id":"tinfoil-kimi-k2-6","name":"kimi-k2-6","created":1784039152,"description":"Tinfoil chat model","context_length":256000,"architecture":{"modality":"text->text+image","input_modalities":["text->text+image"],"output_modalities":["text->text+image"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":1.5150000000000001e-06,"completion":5.3025e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.38784,"max_completion_cost":1.35744,"max_cost":1.35744},"sats_pricing":{"prompt":0.002330650905415571,"completion":0.008157278168954498,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":596.6466317863863,"max_completion_cost":2088.2632112523515,"max_cost":2088.2632112523515},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-kimi-k2-6"},{"id":"tinfoil-glm-5-2","name":"tinfoil-TEE-glm-5-2","created":1784612292,"description":"Tinfoil chat model","context_length":384000,"architecture":{"modality":"text->text","input_modalities":["text->text"],"output_modalities":["text->text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":1.5150000000000001e-06,"completion":5.3025e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.58176,"max_completion_cost":2.0361599999999997,"max_cost":2.0361599999999997},"sats_pricing":{"prompt":0.002330650905415571,"completion":0.008157278168954498,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":894.9699476795794,"max_completion_cost":3132.394816878527,"max_cost":3132.394816878527},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-glm-5-2"},{"id":"tinfoil-gemma4-31b","name":"gemma4-31b","created":1784039181,"description":"Tinfoil chat model","context_length":256000,"architecture":{"modality":"text->text+image","input_modalities":["text->text+image"],"output_modalities":["text->text+image"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":4.04e-07,"completion":1.0099999999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.103424,"max_completion_cost":0.25855999999999996,"max_cost":0.25855999999999996},"sats_pricing":{"prompt":0.0006215069081108189,"completion":0.0015537672702770472,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":159.10576847636966,"max_completion_cost":397.76442119092405,"max_cost":397.76442119092405},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-gemma4-31b"},{"id":"tinfoil-llama3-3-70b","name":"llama3-3-70b","created":1784039231,"description":"Tinfoil chat model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text->text"],"output_modalities":["text->text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":1.7675e-06,"completion":2.7775e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.22624,"max_completion_cost":0.35552,"max_cost":0.35552},"sats_pricing":{"prompt":0.002719092722984833,"completion":0.00427285999326188,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":348.0438685420586,"max_completion_cost":546.9260791375207,"max_cost":546.9260791375207},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-llama3-3-70b"},{"id":"tinfoil-gpt-oss-120b","name":"gpt-oss-120b","created":1784039199,"description":"Tinfoil chat model","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text->text"],"output_modalities":["text->text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":1.515e-07,"completion":6.06e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0198465,"max_completion_cost":0.079386,"max_cost":0.079386},"sats_pricing":{"prompt":0.0002330650905415571,"completion":0.0009322603621662284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.53152686094398,"max_completion_cost":122.12610744377592,"max_cost":122.12610744377592},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-gpt-oss-120b"},{"id":"tinfoil-gpt-oss-safeguard-120b","name":"gpt-oss-safeguard-120b","created":1784039213,"description":"Tinfoil safety model","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text->text"],"output_modalities":["text->text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":1.515e-07,"completion":6.06e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0198465,"max_completion_cost":0.079386,"max_cost":0.079386},"sats_pricing":{"prompt":0.0002330650905415571,"completion":0.0009322603621662284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.53152686094398,"max_completion_cost":122.12610744377592,"max_cost":122.12610744377592},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-gpt-oss-safeguard-120b"},{"id":"nomic-embed-text","name":"nomic-embed-text","created":1713509298,"description":"Tinfoil embedding model","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":5.05e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.000413696,"max_completion_cost":0.0,"max_cost":0.000413696},"sats_pricing":{"prompt":7.768836351385237e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6364230739054786,"max_completion_cost":0.0,"max_cost":0.6364230739054786},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"doc-upload","name":"doc-upload","created":1716076551,"description":"Tinfoil document model","context_length":0,"architecture":{"modality":"text->text+image","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0505,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":5050.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":77.68836351385237,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":7768836.351385237},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"tinfoil-voxtral-small-24b","name":"voxtral-small-24b","created":1784039301,"description":"Tinfoil audio model","context_length":32000,"architecture":{"modality":"text->text","input_modalities":["text->text"],"output_modalities":["text->text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":2.02e-07,"completion":6.06e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.006464,"max_completion_cost":0.019392,"max_cost":0.019392},"sats_pricing":{"prompt":0.00031075345405540947,"completion":0.0009322603621662284,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.944110529773104,"max_completion_cost":29.83233158931931,"max_cost":29.83233158931931},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"tinfoil-voxtral-small-24b"},{"id":"websearch","name":"websearch","created":1738454400,"description":"Tinfoil tool model","context_length":0,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0505,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":5050.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":77.68836351385237,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":7768836.351385237},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"whisper-large-v3-turbo","name":"whisper-large-v3-turbo","created":1716699056,"description":"Tinfoil audio model","context_length":0,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0101,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":1010.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":15.537672702770474,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":1553767.2702770473},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-tts","name":"qwen3-tts","created":1740700800,"description":"Tinfoil tts model","context_length":0,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0101,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":1010.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":15.537672702770474,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":1553767.2702770473},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-tts","name":"voxtral-tts","created":1781900473,"description":"Tinfoil tts model","context_length":0,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.00202,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":202.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":3.1075345405540946,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":310753.4540554095},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-mini-4b-realtime","name":"voxtral-mini-4b-realtime","created":1781900473,"description":"Tinfoil audio model","context_length":0,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.00202,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":202.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":3.1075345405540946,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":310753.4540554095},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"tinfoil","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"sakana-fugu","name":"fugu","created":1784088354,"description":"fugu via sakana-ai-team","context_length":1000000,"architecture":{"modality":"text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"","instruct_type":null},"pricing":{"prompt":5.060099999999999e-06,"completion":3.0310100000000003e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.060099999999999,"max_completion_cost":30.310100000000002,"max_cost":30.310100000000002},"sats_pricing":{"prompt":0.007784374024088006,"completion":0.04662855578101419,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7784.374024088006,"max_completion_cost":46628.55578101419,"max_cost":46628.55578101419},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"generic","canonical_slug":null,"alias_ids":null,"forwarded_model_id":"sakana-fugu"}]}